refactor(site): introduce Site trait and SITES registry

This commit is contained in:
2026-08-14 18:24:03 +08:00
parent 5e23916b40
commit bf4e6159b3
8 changed files with 237 additions and 101 deletions
+22 -1
View File
@@ -1,10 +1,31 @@
use super::model; use super::model;
use crate::media::Media; use crate::media::Media;
use crate::site::{FetchError, Fetched}; use crate::site::{FetchError, Fetched, Site, SiteFuture};
use html_escape::{encode_double_quoted_attribute, encode_text}; use html_escape::{encode_double_quoted_attribute, encode_text};
use regex::Regex; use regex::Regex;
use std::sync::LazyLock; use std::sync::LazyLock;
/// Registry entry for the bluesky adapter (see [`crate::site::Site`]).
pub struct BskySite;
impl Site for BskySite {
fn id(&self) -> &'static str {
"bsky"
}
fn pattern(&self) -> &'static Regex {
&PATTERN
}
fn cache_key(&self, url: &str) -> Option<String> {
cache_key(url)
}
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched> {
Box::pin(async move { fetch_from_url(url).await })
}
}
pub static PATTERN: LazyLock<Regex> = LazyLock::new(|| { pub static PATTERN: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"^(?:https?://)?bsky\.app/profile/([\w.\-:]+)/post/([\w.\-~]+)").unwrap() Regex::new(r"^(?:https?://)?bsky\.app/profile/([\w.\-:]+)/post/([\w.\-~]+)").unwrap()
}); });
+1 -1
View File
@@ -2,5 +2,5 @@ mod interface;
mod model; mod model;
pub use interface::{ pub use interface::{
PATTERN, Post, cache_key, enabled, fetch_from_url, is_retryable, media_headers, BskySite, PATTERN, Post, cache_key, enabled, fetch_from_url, is_retryable, media_headers,
}; };
+129 -82
View File
@@ -5,10 +5,14 @@
//! adding one guarded entry in [`fetch_once`]. //! adding one guarded entry in [`fetch_once`].
use std::fmt; use std::fmt;
use std::future::Future;
use std::pin::Pin;
use std::sync::LazyLock; use std::sync::LazyLock;
use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::atomic::{AtomicBool, Ordering};
use std::time::Duration; use std::time::Duration;
use regex::Regex;
pub mod bsky; pub mod bsky;
pub mod pixiv; pub mod pixiv;
pub mod twitter; pub mod twitter;
@@ -160,17 +164,10 @@ pub fn caption_from_fields(
/// Stable per-post cache key derived from any supported URL, so variant /// Stable per-post cache key derived from any supported URL, so variant
/// domains (x.com / twitter.com / fxtwitter.com, mobile, `/photo/N` /// domains (x.com / twitter.com / fxtwitter.com, mobile, `/photo/N`
/// suffixes) map to the same post. Delegates to the per-site `cache_key` /// suffixes) map to the same post. Delegates to each registered site's
/// implementations (dispatch order twitter → bsky → pixiv). /// `cache_key` (dispatch order twitter → bsky → pixiv).
pub fn cache_key(url: &str) -> Option<String> { pub fn cache_key(url: &str) -> Option<String> {
[ SITES.iter().find_map(|site| site.cache_key(url))
twitter::cache_key(url),
bsky::cache_key(url),
pixiv::cache_key(url),
]
.into_iter()
.flatten()
.next()
} }
/// The site id carried by a cache key (`"twitter:123"` → `"twitter"`). /// The site id carried by a cache key (`"twitter:123"` → `"twitter"`).
@@ -178,18 +175,12 @@ pub fn cache_key(url: &str) -> Option<String> {
/// link-cache hit path, where no [`Fetched`] is available — the same value /// link-cache hit path, where no [`Fetched`] is available — the same value
/// a fresh fetch would read from [`Fetched::site_id`]. /// a fresh fetch would read from [`Fetched::site_id`].
pub fn site_id_from_key(key: &str) -> &'static str { pub fn site_id_from_key(key: &str) -> &'static str {
match key.split(':').next() { let prefix = key.split(':').next().unwrap_or("");
Some("twitter") => "twitter", SITES
Some("pixiv") => "pixiv", .iter()
Some("bsky") => "bsky", .map(|site| site.id())
_ => "unknown", .find(|id| *id == prefix)
} .unwrap_or("unknown")
}
/// Every supported site id, in dispatch order. The bot's SetFormat whitelist
/// and per-site caption-format lookup derive from this list.
pub fn site_ids() -> Vec<&'static str> {
vec!["twitter", "bsky", "pixiv"]
} }
#[derive(Debug)] #[derive(Debug)]
@@ -311,21 +302,110 @@ pub(crate) fn log_once_ffmpeg_missing() {
} }
} }
/// Site adapter: one impl per supported site (twitter / bsky / pixiv),
/// registered in [`SITES`]. All site-specific knowledge — URL pattern,
/// cache-key format, fetch, retry policy, media-host headers, startup
/// validation — lives in the site module; the central dispatcher only
/// iterates the registry.
///
/// Async methods return a boxed future (see [`SiteFuture`]): `async fn` /
/// RPITIT in traits are not dyn-compatible (verified on rustc 1.95), and
/// `+ Send` is required since URL/queue workers spawn these futures. The
/// site structs are stateless unit structs, so the boxed futures never
/// borrow from `self` beyond the call's scope.
pub trait Site: Send + Sync {
/// Stable site id (`"twitter"` / `"bsky"` / `"pixiv"`): caption-format
/// lookup, cache-key prefixes and the SetFormat whitelist derive from it.
fn id(&self) -> &'static str;
/// URL pattern; the dispatcher's first match wins (dispatch order).
fn pattern(&self) -> &'static Regex;
/// Whether the site is usable (env token present, not disabled).
fn enabled(&self) -> bool {
true
}
/// Normalized cache key for a URL of this site (`None` when the URL does
/// not match this site).
fn cache_key(&self, url: &str) -> Option<String>;
/// Fetches and normalizes a post.
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched>;
/// Retry policy for fetch errors: transient classes only.
fn is_retryable(&self, err: &FetchError) -> bool {
matches!(err, FetchError::Http(_) | FetchError::Transient(_))
}
/// Extra headers for downloading this site's media (hotlink protection,
/// e.g. pixiv's Referer for pximg.net). Matched on the media URL, not
/// the site pattern.
fn media_headers(&self, _url: &str) -> Option<Vec<(&'static str, String)>> {
None
}
/// Startup validation (token check etc.); failures are surfaced by
/// [`validate_all`]. The default is a no-op.
fn validate(&self) -> SiteFuture<'static, (), String> {
Box::pin(async { Ok(()) })
}
}
/// A boxed, `Send` future produced by a [`Site`] async method. Boxed so the
/// trait stays dyn-compatible; `Send` because URL/queue workers `tokio::spawn`
/// these futures.
type SiteFuture<'a, T, E = FetchError> = Pin<Box<dyn Future<Output = Result<T, E>> + Send + 'a>>;
/// The one registry of supported sites, in dispatch order (twitter → bsky →
/// pixiv). Adding a site = new module + one `Box::new(...)` entry here; the
/// bot crate never lists sites itself.
static SITES: LazyLock<Vec<Box<dyn Site>>> = LazyLock::new(|| {
vec![
Box::new(twitter::TwitterSite),
Box::new(bsky::BskySite),
Box::new(pixiv::PixivSite),
]
});
/// The first enabled site whose pattern matches `url`, in dispatch order.
fn find_site(url: &str) -> Option<&'static dyn Site> {
SITES
.iter()
.find(|site| site.enabled() && site.pattern().is_match(url))
.map(|site| site.as_ref())
}
/// Every supported site id, in dispatch order. The bot's SetFormat whitelist
/// derives from this list.
pub fn site_ids() -> Vec<&'static str> {
SITES.iter().map(|site| site.id()).collect()
}
/// Runs every enabled site's startup validation and returns the failures
/// (site id + message). The caller logs / notifies; failing sites disable
/// themselves (pixiv disables on a bad token).
pub async fn validate_all() -> Vec<(&'static str, String)> {
let mut failures = Vec::new();
for site in SITES.iter() {
if !site.enabled() {
continue;
}
if let Err(e) = site.validate().await {
failures.push((site.id(), e));
}
}
failures
}
/// Fetches a post from its URL. Returns `Ok(None)` when no site pattern /// Fetches a post from its URL. Returns `Ok(None)` when no site pattern
/// matches (unsupported links are silently ignored by the bot). /// matches (unsupported links are silently ignored by the bot).
/// ///
/// Transient failures are retried: 3 total attempts with 1s then 2s delays. /// Transient failures are retried: 3 total attempts with 1s then 2s delays.
/// What counts as transient is the matched site's own policy /// What counts as transient is the matched site's own policy (`is_retryable`
/// (`SiteKind::is_retryable` — e.g. pixiv retries only network errors and /// — e.g. pixiv retries only network errors and 429/5xx). Permanent classes
/// 429/5xx). Permanent classes (not-found, blocked, sensitive, parse /// (not-found, blocked, sensitive, parse failures, pixiv 4xx/auth errors)
/// failures, pixiv 4xx/auth errors) are returned immediately; retrying them /// are returned immediately; retrying them only wastes attempts against the
/// only wastes attempts against the source site. /// source site.
pub async fn fetch(url: &str) -> Result<Option<Fetched>, FetchError> { pub async fn fetch(url: &str) -> Result<Option<Fetched>, FetchError> {
let Some(site) = match_site(url) else { let Some(site) = find_site(url) else {
return Ok(None); return Ok(None);
}; };
for attempt in 0..3u32 { for attempt in 0..3u32 {
match fetch_once(url, site).await { match site.fetch_from_url(url).await {
Ok(fetched) => { Ok(fetched) => {
// Per-request detail: debug only, keyed by the post id. // Per-request detail: debug only, keyed by the post id.
log::debug!( log::debug!(
@@ -348,63 +428,15 @@ pub async fn fetch(url: &str) -> Result<Option<Fetched>, FetchError> {
unreachable!("retry loop always returns") unreachable!("retry loop always returns")
} }
/// Which site owns a URL (dispatch order twitter → bsky → pixiv), honoring
/// each site's `enabled()` gate. `None` for unsupported links.
fn match_site(url: &str) -> Option<SiteKind> {
if twitter::enabled() && twitter::PATTERN.is_match(url) {
Some(SiteKind::Twitter)
} else if bsky::enabled() && bsky::PATTERN.is_match(url) {
Some(SiteKind::Bsky)
} else if pixiv::enabled() && pixiv::PATTERN.is_match(url) {
Some(SiteKind::Pixiv)
} else {
None
}
}
/// Statically-dispatched site handle: keeps per-site fetch + retry policy
/// callable from the central dispatcher without a trait object (stage 2 of
/// the site-registry refactor; stage 3 replaces this with `dyn Site`).
#[derive(Clone, Copy)]
enum SiteKind {
Twitter,
Bsky,
Pixiv,
}
impl SiteKind {
/// The matched site's own retry policy (see each site's `is_retryable`).
fn is_retryable(&self, err: &FetchError) -> bool {
match self {
SiteKind::Twitter => twitter::is_retryable(err),
SiteKind::Bsky => bsky::is_retryable(err),
SiteKind::Pixiv => pixiv::is_retryable(err),
}
}
}
async fn fetch_once(url: &str, site: SiteKind) -> Result<Fetched, FetchError> {
match site {
SiteKind::Twitter => twitter::fetch_from_url(url).await,
SiteKind::Bsky => bsky::fetch_from_url(url).await,
SiteKind::Pixiv => pixiv::fetch_from_url(url).await,
}
}
/// Applies every site's media-header rule to a download request (pixiv's /// Applies every site's media-header rule to a download request (pixiv's
/// `Referer` for pximg.net hotlink protection). Sites contribute via their /// `Referer` for pximg.net hotlink protection). Sites contribute via their
/// `media_headers(url)` — the central download code carries no per-site logic. /// `media_headers(url)` — the central download code carries no per-site logic.
fn apply_media_headers(mut request: reqwest::RequestBuilder, url: &str) -> reqwest::RequestBuilder { fn apply_media_headers(mut request: reqwest::RequestBuilder, url: &str) -> reqwest::RequestBuilder {
for headers in [ for site in SITES.iter() {
twitter::media_headers(url), if let Some(headers) = site.media_headers(url) {
bsky::media_headers(url), for (name, value) in headers {
pixiv::media_headers(url), request = request.header(name, value);
] }
.into_iter()
.flatten()
{
for (name, value) in headers {
request = request.header(name, value);
} }
} }
request request
@@ -525,6 +557,21 @@ mod tests {
assert_eq!(site_id_from_key("no-colon"), "unknown"); assert_eq!(site_id_from_key("no-colon"), "unknown");
} }
#[test]
fn registry_lists_all_sites_in_dispatch_order() {
assert_eq!(site_ids(), vec!["twitter", "bsky", "pixiv"]);
// Enabled sites dispatch; unsupported URLs never match.
assert!(find_site("https://x.com/u/status/1").is_some());
assert!(find_site("https://bsky.app/profile/u/post/3x").is_some());
assert!(find_site("https://example.com/x").is_none());
// Cache keys are pattern-driven, independent of the enabled() gate
// (pixiv is disabled in tests without PIXIV_REFRESH_TOKEN).
assert_eq!(
cache_key("https://www.pixiv.net/artworks/1"),
Some("pixiv:1".into())
);
}
#[test] #[test]
fn caption_from_fields_substitutes_and_escapes() { fn caption_from_fields_substitutes_and_escapes() {
// The format string is escaped, the field values are substituted // The format string is escaped, the field values are substituted
+48 -1
View File
@@ -1,6 +1,6 @@
use super::model::{IllustrationModel, TypeModel}; use super::model::{IllustrationModel, TypeModel};
use crate::media::Media; use crate::media::Media;
use crate::site::{FetchError, Fetched, PixivError}; use crate::site::{FetchError, Fetched, PixivError, Site, SiteFuture};
use html_escape::{encode_double_quoted_attribute, encode_text}; use html_escape::{encode_double_quoted_attribute, encode_text};
use regex::Regex; use regex::Regex;
use std::sync::LazyLock; use std::sync::LazyLock;
@@ -13,6 +13,53 @@ pub fn enabled() -> bool {
super::api::enabled() super::api::enabled()
} }
/// Registry entry for the pixiv adapter (see [`crate::site::Site`]).
pub struct PixivSite;
impl Site for PixivSite {
fn id(&self) -> &'static str {
"pixiv"
}
fn pattern(&self) -> &'static Regex {
&PATTERN
}
fn enabled(&self) -> bool {
enabled()
}
fn cache_key(&self, url: &str) -> Option<String> {
cache_key(url)
}
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched> {
Box::pin(async move { fetch_from_url(url).await })
}
fn is_retryable(&self, err: &FetchError) -> bool {
is_retryable(err)
}
fn media_headers(&self, url: &str) -> Option<Vec<(&'static str, String)>> {
media_headers(url)
}
fn validate(&self) -> SiteFuture<'static, (), String> {
Box::pin(async {
match super::api::validate().await {
Ok(()) => Ok(()),
Err(e) => {
// Keep the old behavior: a failed login disables pixiv
// for the rest of this process.
super::api::disable();
Err(format!("{e}"))
}
}
})
}
}
pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> { pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> {
let id = PATTERN let id = PATTERN
.captures(url) .captures(url)
+2 -1
View File
@@ -4,5 +4,6 @@ mod model;
pub use api::{PixivAPI, PixivError, disable, fetch, validate}; pub use api::{PixivAPI, PixivError, disable, fetch, validate};
pub use interface::{ pub use interface::{
Illustration, PATTERN, cache_key, enabled, fetch_from_url, is_retryable, media_headers, Illustration, PATTERN, PixivSite, cache_key, enabled, fetch_from_url, is_retryable,
media_headers,
}; };
+22 -1
View File
@@ -1,10 +1,31 @@
use super::model; use super::model;
use crate::media::Media; use crate::media::Media;
use crate::site::{FetchError, Fetched}; use crate::site::{FetchError, Fetched, Site, SiteFuture};
use html_escape::{encode_double_quoted_attribute, encode_text}; use html_escape::{encode_double_quoted_attribute, encode_text};
use regex::Regex; use regex::Regex;
use std::sync::LazyLock; use std::sync::LazyLock;
/// Registry entry for the twitter adapter (see [`crate::site::Site`]).
pub struct TwitterSite;
impl Site for TwitterSite {
fn id(&self) -> &'static str {
"twitter"
}
fn pattern(&self) -> &'static Regex {
&PATTERN
}
fn cache_key(&self, url: &str) -> Option<String> {
cache_key(url)
}
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched> {
Box::pin(async move { fetch_from_url(url).await })
}
}
pub static PATTERN: LazyLock<Regex> = LazyLock::new(|| { pub static PATTERN: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"^(?:https?://)?(?:www\.|mobile\.)?(?:x|twitter|fixvx|vxtwitter|fixupx|fxtwitter)\.com/[^.]+/status/(\d+)").unwrap() Regex::new(r"^(?:https?://)?(?:www\.|mobile\.)?(?:x|twitter|fixvx|vxtwitter|fixupx|fxtwitter)\.com/[^.]+/status/(\d+)").unwrap()
}); });
+1 -1
View File
@@ -3,5 +3,5 @@ mod interface;
mod model; mod model;
pub use interface::{ pub use interface::{
PATTERN, Tweet, cache_key, enabled, fetch_from_url, is_retryable, media_headers, PATTERN, Tweet, TwitterSite, cache_key, enabled, fetch_from_url, is_retryable, media_headers,
}; };
+12 -13
View File
@@ -69,19 +69,18 @@ async fn main() {
handlers::start_url_workers().await; handlers::start_url_workers().await;
log::info!("url workers started"); log::info!("url workers started");
// Pixiv login validation (user request): a failed login notifies the // Site login validation (user request): a failed login notifies the
// admin and disables pixiv for this process. // admin and the site disables itself for this process (pixiv).
if site::pixiv::enabled() { let failures = site::validate_all().await;
match site::pixiv::validate().await { if failures.is_empty() {
Ok(()) => log::info!("pixiv login validated"), log::info!("site logins validated");
Err(e) => { } else {
log::error!("pixiv login failed: {e}"); for (site_id, message) in &failures {
if let Some(admin) = CONFIG.admin_ids.first() { log::error!("{site_id} login failed: {message}");
let _ = bot if let Some(admin) = CONFIG.admin_ids.first() {
.send_message(ChatId(*admin), format!("Pixiv login failed: {e}")) let _ = bot
.await; .send_message(ChatId(*admin), format!("{site_id} login failed: {message}"))
} .await;
site::pixiv::disable();
} }
} }
} }