mirror of
https://github.com/TheFunny/TelegramTwitterMediaBot.git
synced 2026-09-23 23:32:05 +00:00
perf(pixiv): stream the ugoira frame zip to disk instead of RAM
download_media_limited buffered the whole frame zip (cap 512 MB) in memory before extraction, spiking RAM for large ugoira. New site::download_media_to_file streams chunks straight to a temp file with the same Content-Length / stream cap checks, and ugoira_video now opens the zip from disk inside spawn_blocking. Adds FetchError::Io for local write failures (hand-rolled error pattern preserved).
This commit is contained in:
@@ -191,6 +191,9 @@ pub enum FetchError {
|
||||
TooLarge,
|
||||
/// A transient server-side failure (429 / 5xx); [`fetch`] retries these.
|
||||
Transient(String),
|
||||
/// A local I/O failure while streaming a download to disk
|
||||
/// (see [`download_media_to_file`]).
|
||||
Io(std::io::Error),
|
||||
}
|
||||
|
||||
impl fmt::Display for FetchError {
|
||||
@@ -204,6 +207,7 @@ impl fmt::Display for FetchError {
|
||||
FetchError::Sensitive => write!(f, "content withheld (sensitive)"),
|
||||
FetchError::TooLarge => write!(f, "media too large"),
|
||||
FetchError::Transient(message) => write!(f, "transient: {message}"),
|
||||
FetchError::Io(e) => write!(f, "io error: {e}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -217,6 +221,7 @@ impl std::error::Error for FetchError {
|
||||
FetchError::NotFound | FetchError::Blocked | FetchError::Sensitive => None,
|
||||
FetchError::TooLarge => None,
|
||||
FetchError::Transient(_) => None,
|
||||
FetchError::Io(e) => Some(e),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -384,6 +389,41 @@ pub async fn download_media(url: &str) -> Result<bytes::Bytes, FetchError> {
|
||||
download_media_limited(url, u64::MAX).await
|
||||
}
|
||||
|
||||
/// Streams a download to `out`, aborting with [`FetchError::TooLarge`] the
|
||||
/// moment the body crosses `max_bytes` (or when a declared Content-Length
|
||||
/// already exceeds it). Unlike [`download_media_limited`] the body is never
|
||||
/// buffered in memory — used for large files (e.g. the pixiv ugoira frame
|
||||
/// zip, which can be hundreds of MB) that would otherwise spike RAM.
|
||||
/// Returns the number of bytes written.
|
||||
pub async fn download_media_to_file(
|
||||
url: &str,
|
||||
max_bytes: u64,
|
||||
out: &mut std::fs::File,
|
||||
) -> Result<u64, FetchError> {
|
||||
use std::io::Write;
|
||||
let mut request = CLIENT.get(url);
|
||||
let lower = url.to_ascii_lowercase();
|
||||
if lower.contains("pximg.net") {
|
||||
request = request.header("Referer", "https://www.pixiv.net/");
|
||||
}
|
||||
let response = request.send().await?.error_for_status()?;
|
||||
if let Some(len) = response.content_length()
|
||||
&& len > max_bytes
|
||||
{
|
||||
return Err(FetchError::TooLarge);
|
||||
}
|
||||
let mut response = response;
|
||||
let mut total: u64 = 0;
|
||||
while let Some(chunk) = response.chunk().await? {
|
||||
total += chunk.len() as u64;
|
||||
if total > max_bytes {
|
||||
return Err(FetchError::TooLarge);
|
||||
}
|
||||
out.write_all(&chunk).map_err(FetchError::Io)?;
|
||||
}
|
||||
Ok(total)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -449,7 +489,11 @@ mod tests {
|
||||
fn truncate_caption_cuts_long_text_with_ellipsis() {
|
||||
let long = "x".repeat(MAX_CAPTION_CHARS + 100);
|
||||
let out = truncate_caption(&long);
|
||||
assert!(out.chars().count() <= MAX_CAPTION_CHARS, "len {}", out.chars().count());
|
||||
assert!(
|
||||
out.chars().count() <= MAX_CAPTION_CHARS,
|
||||
"len {}",
|
||||
out.chars().count()
|
||||
);
|
||||
assert!(out.ends_with('…'));
|
||||
}
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ use crate::media::Media;
|
||||
use crate::site::FetchError;
|
||||
use std::env;
|
||||
use std::fmt;
|
||||
use std::io::{Cursor, Read};
|
||||
use std::io::Read;
|
||||
use std::sync::LazyLock;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::time::{Duration, SystemTime};
|
||||
@@ -228,7 +228,14 @@ impl PixivAPI {
|
||||
let Some(zip_url) = zip_url else {
|
||||
return Ok(None);
|
||||
};
|
||||
let zip_bytes = crate::site::download_media_limited(&zip_url, 512 * 1024 * 1024)
|
||||
// Stream the frame zip to a temp file instead of buffering it in
|
||||
// memory: ugoira zips can be hundreds of MB, and the old
|
||||
// download_media_limited path spiked RAM up to the size cap.
|
||||
let mut zip_file = tempfile::Builder::new()
|
||||
.suffix(".zip")
|
||||
.tempfile()
|
||||
.map_err(|e| PixivError::Api(format!("temp zip failed: {e}")))?;
|
||||
crate::site::download_media_to_file(&zip_url, 512 * 1024 * 1024, zip_file.as_file_mut())
|
||||
.await
|
||||
.map_err(|e| match e {
|
||||
FetchError::Http(e) => PixivError::Http(e),
|
||||
@@ -241,9 +248,12 @@ impl PixivAPI {
|
||||
let out_dir = tempfile::tempdir().map_err(|e| e.to_string())?;
|
||||
|
||||
// Extract frames to canonical zero-padded names; pixiv ugoira
|
||||
// frames are uniformly jpg or png per artwork.
|
||||
let mut archive = zip::ZipArchive::new(Cursor::new(zip_bytes))
|
||||
.map_err(|e| format!("unzip: {e}"))?;
|
||||
// frames are uniformly jpg or png per artwork. The zip is read
|
||||
// from disk; `zip_file` stays alive for the whole extraction.
|
||||
let mut archive = zip::ZipArchive::new(
|
||||
std::fs::File::open(zip_file.path()).map_err(|e| e.to_string())?,
|
||||
)
|
||||
.map_err(|e| format!("unzip: {e}"))?;
|
||||
if archive.is_empty() {
|
||||
return Err("empty frame zip".to_string());
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user