mirror of
https://github.com/TheFunny/TelegramTwitterMediaBot.git
synced 2026-09-23 23:32:05 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
65a9554173
|
||
|
|
e755785147
|
||
|
|
5d5b6d56e7
|
||
|
|
d0fdf1c5da
|
||
|
|
1db4ecfafa
|
||
|
|
47039bbe2d
|
||
|
|
bc954e6e0b
|
||
|
|
f7cb809e5a
|
||
|
|
51b40cdb42
|
||
|
|
ab2306002a
|
||
|
|
a92b12f633
|
||
|
|
74e3b7593c
|
||
|
|
2faccaac42
|
||
|
|
f4e60d8946
|
||
|
|
3006dcd98c
|
||
|
|
8e8acdd859
|
||
|
|
8b9dd963e1
|
||
|
|
e83d48f1f7
|
||
|
|
7c55b26731
|
||
|
|
950db48a13
|
||
|
|
32ea8ec6ca
|
||
|
|
62dc033452
|
||
|
|
2801aaa39c
|
||
|
|
93d47752cf
|
||
|
|
6fd4edb3c5
|
||
|
|
7c5afce0b4
|
@@ -28,5 +28,9 @@ LICENSE
|
||||
README.md
|
||||
data/
|
||||
cert/
|
||||
nginx-certs/
|
||||
nginx-vhost.d/
|
||||
nginx-html/
|
||||
nginx-acme/
|
||||
**/target/
|
||||
.idea/
|
||||
|
||||
@@ -12,12 +12,35 @@ env:
|
||||
DOCKERHUB_REPO: yoursfunny/telegram-twitter-media-bot
|
||||
|
||||
jobs:
|
||||
# A tag push and a branch push to the same commit fire two workflow runs;
|
||||
# build only once. Tag runs always build; master runs build only when the
|
||||
# pushed commit is not already tagged (the tag run covers it).
|
||||
should-build:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
build: ${{ steps.check.outputs.build }}
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- id: check
|
||||
shell: bash
|
||||
run: |
|
||||
if [ "$GITHUB_REF_TYPE" = "branch" ] && git tag --points-at "$GITHUB_SHA" | grep -q .; then
|
||||
echo "commit already tagged; the tag run builds the image"
|
||||
echo "build=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "build=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
docker:
|
||||
needs: should-build
|
||||
if: needs.should-build.outputs.build == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Docker meta
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
uses: docker/metadata-action@v6
|
||||
with:
|
||||
images: ${{ env.DOCKERHUB_REPO }}
|
||||
tags: |
|
||||
@@ -28,22 +51,30 @@ jobs:
|
||||
type=sha
|
||||
-
|
||||
name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
uses: docker/setup-qemu-action@v4
|
||||
-
|
||||
name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@v4
|
||||
-
|
||||
name: Login to Docker Hub
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
# Buildkit cache via the GitHub Actions cache backend (uses the
|
||||
# automatic GITHUB_TOKEN, no extra secrets). mode=max keeps every
|
||||
# stage's layers so the cargo-deps and ffmpeg layers are restored
|
||||
# instead of re-downloaded/recompiled. The scope must be pinned to a
|
||||
# fixed string: the gha backend defaults to the current git ref, which
|
||||
# would give every new tag a cold cache on release builds.
|
||||
-
|
||||
name: Build and push
|
||||
uses: docker/build-push-action@v5
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
push: true
|
||||
build-args: |
|
||||
APP_NAME=${{ env.APP_NAME }}
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
cache-from: type=gha,scope=tgxmb-build
|
||||
cache-to: type=gha,mode=max,scope=tgxmb-build
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
__pycache__/
|
||||
cert/
|
||||
data/
|
||||
nginx-certs/
|
||||
nginx-vhost.d/
|
||||
nginx-html/
|
||||
nginx-acme/
|
||||
docker-compose.yml
|
||||
|
||||
.env
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
# Repository Guidelines
|
||||
|
||||
## Project Overview
|
||||
|
||||
Telegram bot (teloxide) that turns post links from X/Twitter, Pixiv, and Bluesky into media messages (images, video, GIF) with the post's title, author, and tags. It supports batch media splitting, retry with persistence, inline queries, forward-channel rebinding with caption templates, and Pixiv ugoira→MP4 transcoding. README and user-facing strings are in Chinese. The project is a Rust port of a Python predecessor (see `queue.rs` comments referencing `utils/task_queue.py`).
|
||||
|
||||
Two-crate Cargo workspace (both v1.0.3, edition 2024, resolver 3):
|
||||
|
||||
- **`crates/x-media`** — library that fetches and normalizes media from the three sites. Pure, no Telegram knowledge.
|
||||
- **`crates/xmedia-bot`** — the bot binary: teloxide dispatcher, SQLite-backed chat state, persistent task queue.
|
||||
|
||||
## Architecture & Data Flow
|
||||
|
||||
```
|
||||
Telegram update → Dispatcher (polling or axum webhook) → dptree branches
|
||||
├─ message → commands (any chat) / URL links (private chat only)
|
||||
├─ inline_query → InlineQueryResult Photo/Video/Mpeg4Gif
|
||||
└─ callback_query → "forward" (copy to channel) / "template|<name>" (apply caption template)
|
||||
```
|
||||
|
||||
Message flow: `message_handler` extracts URLs (from `url`/`text_link` entities, text + caption, deduped) → `x_media::site::fetch(url)` → `Fetched` → builds a `Task` → `send::send_media_sequence` (media groups ≤ 9, caption on first item) or `send::send_animation`. On Telegram URL-fetch failure or size error (`send_batch_via_upload`): download via `x_media::site::download_media` to a temp file (≤ 10 MiB), sniff magic bytes (`sniff_ext`), upload via multipart; oversized items fall back to `fallback_url`. On failure: `enqueue_retry` persists resume-state `Task` into the SQLite queue → single worker leases (120 s lock TTL) → retry with exponential backoff (≤ 30 s, `MAX_RETRIES = 2`) → dead-letter → `notify_failure`. Success → `post_send_actions`: edit-before-forward prompt with inline buttons, or `copy_messages` to the bound forward channel.
|
||||
|
||||
The `x-media` library: `site::fetch(url)` dispatches (in order) twitter → bsky → pixiv via per-site regex `PATTERN` and returns `Ok(None)` for unmatched URLs. `Fetched { source_url, caption, title, media: Vec<Media>, sensitive, … }`; `caption_with(format)` substitutes `{url} {author} {author_url} {title} {tags}`.
|
||||
|
||||
## Key Directories
|
||||
|
||||
| Path | Purpose |
|
||||
|---|---|
|
||||
| `crates/x-media/src/` | Fetch library. `site/mod.rs` = dispatcher + `Fetched`/`FetchError`/`download_media`/`media_size`; `media.rs` = `Media` enum; `examples/fetch.rs` = end-to-end usage sample |
|
||||
| `crates/x-media/src/site/<twitter\|pixiv\|bsky>/` | One directory per site: `mod.rs` (re-exports), `interface.rs` (PATTERN, `enabled()`, `fetch_from_url()`, site struct, `From<SiteStruct> for Fetched`), `model.rs` (serde DTOs). Pixiv adds `api.rs` (auth + transport); twitter adds `auth.rs` (logged-in GraphQL `TweetDetail` fallback for NSFW tweets, gated on `TWITTER_AUTH_TOKEN`) |
|
||||
| `crates/xmedia-bot/src/main.rs` | Entry point: env/log init, queue worker start, pixiv validation, 300 s edit-expiry sweep, dptree handler tree, webhook vs polling dispatch |
|
||||
| `crates/xmedia-bot/src/config.rs` | Manual env parsing into `Config` |
|
||||
| `crates/xmedia-bot/src/handlers.rs` | `Command` enum (teloxide `BotCommands`), message/inline/callback handlers, URL extraction, global statics; per-URL work spawned with a `Semaphore(8)` cap (teloxide's per-chat workers are sequential — batch-forwards need concurrency) |
|
||||
| `crates/xmedia-bot/src/state.rs` | `ChatStore`: parking_lot `Mutex<HashMap>` cache + SQLite write-through (`chat_state` table) |
|
||||
| `crates/xmedia-bot/src/link_cache.rs` | `LinkCache`: SQLite-backed cache (`link_cache` table) of successfully sent posts — raw caption fields + Telegram `file_id`s; repeat links re-send locally (no fetch/upload), TTL + prune, invalidated on permanent send failure |
|
||||
| `crates/xmedia-bot/src/queue.rs` | `PersistentTaskQueue`: SQLite-backed queue (`tasks` table), `QUEUE_WORKERS = 4` concurrent workers (lease via `BEGIN IMMEDIATE` + `locked_until` TTL), retry→dead-letter, `Notify::notify_waiters` wakeup, `busy_timeout` on all connections |
|
||||
| `crates/xmedia-bot/src/send.rs` | Media senders, upload fallback, error classification, queue task handlers |
|
||||
|
||||
## Development Commands
|
||||
|
||||
```bash
|
||||
export TELOXIDE_TOKEN=<token> # required; PIXIV_REFRESH_TOKEN optional (Pixiv disabled without it)
|
||||
cargo run -p xmedia-bot # run the bot (polling by default)
|
||||
cargo run -p x-media --example fetch -- <url> # test a link through the fetch library
|
||||
cargo test --workspace # full test suite (no CI test step exists — run locally)
|
||||
cargo build --release -p xmedia-bot # release build (Dockerfile does this)
|
||||
cargo clippy --workspace --all-targets # lint (Clippy is the configured IDE linter)
|
||||
cargo fmt --check # formatting
|
||||
```
|
||||
|
||||
Docker: `docker build -t tgxmb .` then `docker run --rm -d --name tgxmb --env-file .env -v ./data:/app/data tgxmb`. Runtime requires **ffmpeg** (built into the image).
|
||||
|
||||
## Code Conventions & Common Patterns
|
||||
|
||||
- **No anyhow/thiserror.** Errors are hand-rolled enums with manual `Display`/`source()`/`From` impls: `QueueError` (`Retryable { delay_seconds, payload }` / `Permanent`), `SendError` (Retryable/Permanent), `FetchError` (`Http`/`Json`/`Pixiv`/`NotFound`/`Blocked`), `PixivError`, `Classification`. New errors should follow this pattern.
|
||||
- **Global state via `std::sync::LazyLock` statics**, not DI: `CONFIG`, `CHAT_STORE`, `TASK_QUEUE` in `handlers.rs`; shared reqwest `CLIENT` in `x-media/src/site/mod.rs`. `Bot` is passed/cloned into handlers; queue workers rebuild `Bot::from_env()`.
|
||||
- **Async**: tokio multi-thread runtime (`#[tokio::main]` default). All rusqlite I/O inside `tokio::task::spawn_blocking`. Long loops use `tokio::select!` with `tokio::sync::{watch, Notify}` stop/wake channels. No streams.
|
||||
- **Blocking sync primitives**: `parking_lot::Mutex` for hot caches, `tokio::sync::Mutex` for async-shared state (pixiv token cache), `AtomicBool` for feature gates.
|
||||
- **Site adapter convention** (no trait, no enum dispatch — follow the existing convention): each site module exports `PATTERN: LazyLock<Regex>`, `enabled() -> bool`, `fetch_from_url(url) -> Result<Fetched, FetchError>`; `site/mod.rs` re-exports the site struct and `fetch_once` adds one guarded if-branch. Adding a site = new `site/<name>/{mod.rs,interface.rs,model.rs}` + one branch in `fetch_once`.
|
||||
- **Serde**: per-site `model.rs` are pure `Deserialize` DTOs mirroring API JSON; site structs in `interface.rs` have private fields, a `caption()` builder, and `impl From<SiteStruct> for Fetched`. Persisted payloads use internally-tagged enums (`#[serde(tag = "kind")]` / `type`).
|
||||
- **Naming**: module-per-concern, snake_case files, `CamelCase` types, `snake_case` fns. `//!` module docs and `///` docs on non-obvious logic (syndication token, ugoira encoding, `display_text_range`).
|
||||
- **Retries**: only `x-media::site::fetch` retries (3 attempts, `1 << attempt` backoff, HTTP errors only). Queue retries are explicit `QueueError::Retryable` with computed delay (`retry_delay_seconds`).
|
||||
- Logging via `log` macros (`pretty_env_logger`, level from `RUST_LOG`).
|
||||
|
||||
## Important Files
|
||||
|
||||
| File | Why it matters |
|
||||
|---|---|
|
||||
| `crates/xmedia-bot/src/main.rs` | Startup sequence, webhook vs polling, graceful shutdown (SIGINT via teloxide ctrlc / SIGTERM via `stop_token` for docker, → sweep stop → admin msg → queue stop) |
|
||||
| `crates/xmedia-bot/src/handlers.rs` | `CHAT_STORE`/`TASK_QUEUE`/`CONFIG` singletons (open `data/task_queue.db` **relative to CWD**); command dispatch; URL extraction; retry enqueue |
|
||||
| `crates/xmedia-bot/src/send.rs` | Constants `MAX_MEDIA_GROUP = 9`; fallback chain; `classify_request_error`; download-and-reupload fallback triggered only by Telegram API errors (`is_media_fetch_failure` / `is_size_error`) |
|
||||
| `crates/xmedia-bot/src/photo.rs` | Pure-Rust photo processing (no ffmpeg): `png` (image-png) decode/encode + `zune-jpeg` decode + `fast_image_resize` Lanczos3 downscale + `jpeg-encoder`. Photos over Telegram's limits (width + height > 10000 px → `PHOTO_INVALID_DIMENSIONS`; bytes > 10 MiB) are decoded, downscaled keeping the format, PNG bit depth > 24 (RGBA 32-bit / 16-bit per channel) reduced to 24-bit RGB with alpha flattened white (≤24-bit untouched, never upconverted), and transcoded to JPEG only if still over the cap; memory budget guarded, otherwise the item's smaller fallback URL |
|
||||
| `crates/x-media/src/site/mod.rs` | Dispatcher, `Fetched`/`FetchError`, shared `CLIENT`, `download_media` (adds `Referer: https://www.pixiv.net/` for `pximg.net` hotlink protection) |
|
||||
| `crates/x-media/src/site/pixiv/api.rs` | OAuth token exchange (hardcoded app client id/secret), access-token cache, ugoira zip→MP4 via ffmpeg in `spawn_blocking` |
|
||||
| `Dockerfile` | Multi-stage: cached dep layer via stub sources + `touch *.rs` mtime hack, static ffmpeg from ffmpeg.martin-riedl.de (`FFMPEG_URL` arg, `unzip -t` integrity check), `debian:bookworm-slim` runtime, entrypoint |
|
||||
| `docker-entrypoint.sh` | Privilege drop: `useradd` with `LOCAL_USER_ID` (default 9001) + `setpriv` (no gosu on bookworm-slim) |
|
||||
| `docker-compose.yml.example` | Deployment env reference (real `docker-compose.yml` is gitignored). Ships nginx-proxy + acme-companion: webhook mode needs TLS termination in front (teloxide's axum listener is HTTP-only; `WEBHOOK_CERT` only feeds `set_webhook`), bot exposes `VIRTUAL_HOST`/`VIRTUAL_PORT` on the shared `proxy` network, no host port; container names `nginx-proxy`/`acme-companion`/`tgxmb`, start order via `depends_on` (proxy → acme → bot) |
|
||||
| `.github/workflows/docker.yml` | CI: build+push to Docker Hub on tag `v*`/master; **no test step**; buildx gha cache (`cache-from`/`cache-to`, scope `tgxmb-build`, `mode=max`) so cargo deps + ffmpeg layers are restored across runs |
|
||||
| `README.md` | Feature docs + command table (Chinese) |
|
||||
|
||||
## Runtime/Tooling Preferences
|
||||
|
||||
- **Rust, stable, edition 2024**, workspace resolver 3. No `rust-version`/MSRV pin, no `rust-toolchain.toml` — recent stable is assumed. No nightly features.
|
||||
- Package manager: **Cargo** (workspace with path dep `x-media` ← `xmedia-bot`). No `[workspace.package]`/shared deps — each crate lists deps independently.
|
||||
- **Two reqwest versions coexist in the lock** (0.12.28 via teloxide, 0.13.3 in x-media) — don't unify casually.
|
||||
- Config is **environment-variable driven** (dotenv loads `.env`, gitignored; no `.env.example` exists). Key vars: `TELOXIDE_TOKEN` (required), `PIXIV_REFRESH_TOKEN`, `TWITTER_AUTH_TOKEN` (optional; x.com `auth_token` cookie — enables the logged-in GraphQL fallback that fetches NSFW tweets syndication withholds), `BOT_ADMIN` (comma-separated ids), `EDIT_MESSAGE_TTL_SECONDS` (default 86400), `LINK_CACHE_TTL_SECONDS` (default 604800), `WEBHOOK`/`WEBHOOK_URL`/`WEBHOOK_LISTEN`/`WEBHOOK_PORT`/`WEBHOOK_CERT`/`WEBHOOK_SECRET_TOKEN` (webhook mode requires URL/listen/port, `.expect`ed; `WEBHOOK_CERT` is Telegram-facing self-signed validation only — TLS must be terminated by a reverse proxy), `RUST_LOG`, `TELOXIDE_PROXY`, `LOCAL_USER_ID` (entrypoint only).
|
||||
- SQLite via `rusqlite` with `bundled` feature (no system libsqlite needed). DB file `data/task_queue.db` is CWD-relative — run from the workspace root, or `/app` in Docker. Mount `./data` and `./cert` volumes.
|
||||
- `.gitattributes` enforces LF for `*.sh` (CRLF breaks shebangs in containers). `.gitignore`: `.env`, `data/`, `cert/`, `docker-compose.yml`, `/target`, `.idea/`.
|
||||
- Docs are in Chinese; user-facing bot strings too. Keep that convention when editing captions/templates/docs.
|
||||
|
||||
## Testing & QA
|
||||
|
||||
- **~51 tests, all inline `#[cfg(test)] mod tests`** — no `tests/` integration directories. Framework: built-in Rust test + `#[tokio::test]` (dev-deps only in `x-media`: tokio macros/rt-multi-thread, dotenv).
|
||||
- No mocking framework anywhere (no mockito/wiremock/mockall). Conventions: pure-function units (regex parsing, serde round-trips, chunking, retry math) tested synchronously; async tests use real dependencies — file-backed SQLite via `tempfile` (`queue.rs::new_queue()` helper), live network fetches.
|
||||
- Live-network tests exist in `site/twitter/interface.rs` (3), `site/bsky/interface.rs` (2), `site/pixiv/interface.rs`/`api.rs` (env-gated on `PIXIV_REFRESH_TOKEN`/dotenv, skip by early return). Run the full suite with `cargo test --workspace`.
|
||||
- Fixtures are inline `serde_json::json!` builder fns (`fixture()`, `thread_json()`, `illust_json()`), not files. The shared `CLIENT` sets `pool_max_idle_per_host(0)` under `#[cfg(test)]` to avoid cross-runtime `DispatchGone`.
|
||||
- **CI runs no tests** — `.github/workflows/docker.yml` only builds/pushes the image; verification is a local responsibility.
|
||||
- Untested and hard to test without a mock seam: `handlers.rs` (depends directly on teloxide `Bot`); `main.rs`, `config.rs`, `state.rs`; `media.rs`, `lib.rs`, all `model.rs`.
|
||||
- No coverage tracking, no lint gate in CI.
|
||||
Generated
+77
-2
@@ -505,6 +505,15 @@ dependencies = [
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "document-features"
|
||||
version = "0.2.12"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d4b8a88685455ed29a21542a33abd9cb6510b6b129abadabdcef0f4c55bc8f61"
|
||||
dependencies = [
|
||||
"litrs",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "dotenv"
|
||||
version = "0.15.0"
|
||||
@@ -599,12 +608,33 @@ version = "0.1.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a"
|
||||
|
||||
[[package]]
|
||||
name = "fast_image_resize"
|
||||
version = "6.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e9c50201dc184ba6553da1695aac20a042efffbe2d84542cee31917c86c3ab1e"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"document-features",
|
||||
"num-traits",
|
||||
"thiserror",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "fastrand"
|
||||
version = "2.4.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6"
|
||||
|
||||
[[package]]
|
||||
name = "fdeflate"
|
||||
version = "0.3.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e6853b52649d4ac5c0bd02320cddc5ba956bdb407c4b75a2c6b75bf51500f8c"
|
||||
dependencies = [
|
||||
"simd-adler32",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "find-msvc-tools"
|
||||
version = "0.1.9"
|
||||
@@ -1305,6 +1335,12 @@ dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "jpeg-encoder"
|
||||
version = "0.7.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a0370574b86f7eca156b9f298392b5e69a23f8c86f3f865add60bbc2e79467a6"
|
||||
|
||||
[[package]]
|
||||
name = "js-sys"
|
||||
version = "0.3.98"
|
||||
@@ -1352,6 +1388,12 @@ version = "0.8.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0"
|
||||
|
||||
[[package]]
|
||||
name = "litrs"
|
||||
version = "1.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "11d3d7f243d5c5a8b9bb5d6dd2b1602c0cb0b9db1621bafc7ed66e35ff9fe092"
|
||||
|
||||
[[package]]
|
||||
name = "lock_api"
|
||||
version = "0.4.14"
|
||||
@@ -1604,6 +1646,19 @@ version = "0.3.33"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e"
|
||||
|
||||
[[package]]
|
||||
name = "png"
|
||||
version = "0.18.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "60769b8b31b2a9f263dae2776c37b1b28ae246943cf719eb6946a1db05128a61"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"crc32fast",
|
||||
"fdeflate",
|
||||
"flate2",
|
||||
"miniz_oxide",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "potential_utf"
|
||||
version = "0.1.5"
|
||||
@@ -3296,12 +3351,13 @@ checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4"
|
||||
|
||||
[[package]]
|
||||
name = "x-media"
|
||||
version = "1.0.2"
|
||||
version = "1.0.7"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"dotenv",
|
||||
"html-escape",
|
||||
"log",
|
||||
"rand 0.8.6",
|
||||
"regex",
|
||||
"reqwest 0.13.3",
|
||||
"serde",
|
||||
@@ -3314,12 +3370,15 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "xmedia-bot"
|
||||
version = "1.0.2"
|
||||
version = "1.0.7"
|
||||
dependencies = [
|
||||
"dotenv",
|
||||
"fast_image_resize",
|
||||
"html-escape",
|
||||
"jpeg-encoder",
|
||||
"log",
|
||||
"parking_lot",
|
||||
"png",
|
||||
"pretty_env_logger",
|
||||
"rand 0.8.6",
|
||||
"regex",
|
||||
@@ -3331,6 +3390,7 @@ dependencies = [
|
||||
"tokio",
|
||||
"url",
|
||||
"x-media",
|
||||
"zune-jpeg",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -3534,3 +3594,18 @@ dependencies = [
|
||||
"cc",
|
||||
"pkg-config",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zune-core"
|
||||
version = "0.5.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cb8a0807f7c01457d0379ba880ba6322660448ddebc890ce29bb64da71fb40f9"
|
||||
|
||||
[[package]]
|
||||
name = "zune-jpeg"
|
||||
version = "0.5.15"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "27bc9d5b815bc103f142aa054f561d9187d191692ec7c2d1e2b4737f8dbd7296"
|
||||
dependencies = [
|
||||
"zune-core",
|
||||
]
|
||||
|
||||
+16
-9
@@ -1,12 +1,16 @@
|
||||
# ---------- build stage ----------
|
||||
# rust:1-bookworm (full, not slim) ships the C toolchain needed by
|
||||
# rusqlite's bundled SQLite, plus wget/xz for the ffmpeg download.
|
||||
# rusqlite's bundled SQLite, plus wget/unzip for the ffmpeg download.
|
||||
FROM rust:1-bookworm AS builder
|
||||
|
||||
ARG APP_NAME=telegram-twitter-media-bot
|
||||
# Statically compiled ffmpeg (ugoira MP4 encoding). amd64 by default; override
|
||||
# for other platforms or pin a different johnvansickle build.
|
||||
ARG FFMPEG_URL=https://johnvansickle.com/ffmpeg/releases/ffmpeg-7.0.2-amd64-static.tar.xz
|
||||
# Prebuilt static ffmpeg (glibc-linked, includes libx264) for ugoira MP4
|
||||
# encoding. Served from https://ffmpeg.martin-riedl.de (Cloudflare CDN,
|
||||
# built on Debian 12 — glibc-compatible with the bookworm-slim runtime).
|
||||
# johnvansickle.com throttles datacenter IPs and served garbage from GitHub
|
||||
# runners. `/redirect/latest/` floats to the newest release build; each build
|
||||
# also ships a .sha256. Swap `amd64` for `arm64` when building arm64 images.
|
||||
ARG FFMPEG_URL=https://ffmpeg.martin-riedl.de/redirect/latest/linux/amd64/release/ffmpeg.zip
|
||||
|
||||
WORKDIR /build
|
||||
|
||||
@@ -22,11 +26,14 @@ RUN mkdir -p crates/x-media/src crates/xmedia-bot/src \
|
||||
&& cargo build --release -p xmedia-bot
|
||||
|
||||
# 2. Static ffmpeg next (cached unless FFMPEG_URL changes), so source edits
|
||||
# never re-download it. The johnvansickle tarball has a
|
||||
# `{build}/ffmpeg` layout, so strip one path component.
|
||||
RUN wget -q -O /tmp/ffmpeg.tar.xz "$FFMPEG_URL" \
|
||||
&& tar -xJf /tmp/ffmpeg.tar.xz -C /usr/local/bin --strip-components=1 --wildcards '*/ffmpeg' \
|
||||
&& rm /tmp/ffmpeg.tar.xz \
|
||||
# never re-download it. The zip contains a single `ffmpeg` binary at the
|
||||
# root. `unzip -t` verifies the archive before extraction so a bad
|
||||
# download fails loudly here instead of a cryptic later error.
|
||||
RUN wget -q -O /tmp/ffmpeg.zip "$FFMPEG_URL" \
|
||||
&& unzip -tq /tmp/ffmpeg.zip \
|
||||
&& unzip -q /tmp/ffmpeg.zip -d /usr/local/bin \
|
||||
&& chmod +x /usr/local/bin/ffmpeg \
|
||||
&& rm /tmp/ffmpeg.zip \
|
||||
&& /usr/local/bin/ffmpeg -version >/dev/null
|
||||
|
||||
# 3. Real sources last: only our crates recompile on source changes. The
|
||||
|
||||
@@ -10,6 +10,7 @@ Telegram 机器人,将 X / Twitter、Pixiv、Bluesky 的帖子链接转换为
|
||||
- 可绑定转发频道自动转发;支持转发前编辑 caption 与自定义模板
|
||||
- 发送失败自动重试并持久化,重试耗尽后通知用户
|
||||
- Pixiv ugoira 动图自动转码为 MP4
|
||||
- 链接结果本地缓存:成功发送后缓存 Telegram file id 与 caption 等,再次收到相同链接直接本地重发,不再请求源站、不保存媒体文件(`LINK_CACHE_TTL_SECONDS` 控制过期,默认 7 天)
|
||||
|
||||
## 快速开始
|
||||
|
||||
@@ -28,7 +29,73 @@ docker build -t tgxmb .
|
||||
docker run --rm -d --name tgxmb --env-file .env -v ./data:/app/data tgxmb
|
||||
```
|
||||
|
||||
环境变量:`TELOXIDE_TOKEN`(必填)、`PIXIV_REFRESH_TOKEN`、`BOT_ADMIN`、`EDIT_MESSAGE_TTL_SECONDS`、`RUST_LOG`、`WEBHOOK*`。
|
||||
环境变量:`TELOXIDE_TOKEN`(必填)、`PIXIV_REFRESH_TOKEN`、`BOT_ADMIN`、`EDIT_MESSAGE_TTL_SECONDS`、`LINK_CACHE_TTL_SECONDS`、`RUST_LOG`、`WEBHOOK*`、`TWITTER_AUTH_TOKEN`(可选)。
|
||||
|
||||
NSFW 推文:公开的 syndication 接口不返回敏感内容。设置 `TWITTER_AUTH_TOKEN`(登录 x.com 后浏览器 Cookie 里的 `auth_token` 值)后,bot 会仅在遇到 NSFW 推文时以登录态获取媒体;未设置则提示无媒体。
|
||||
|
||||
### Webhook 部署(需要反向代理)
|
||||
|
||||
`docker-compose.yml.example` 内置了 [nginx-proxy](https://github.com/nginx-proxy/nginx-proxy) + [acme-companion](https://github.com/nginx-proxy/acme-companion) 反向代理编排,按部署环境二选一:
|
||||
|
||||
**有域名**
|
||||
1. DNS A 记录指向服务器
|
||||
2. compose 里设 `VIRTUAL_HOST`、`WEBHOOK_URL` 为域名,并取消注释 `ACME_HOST`(设为域名)
|
||||
3. acme-companion 自动签发与续期证书,无需手动处理
|
||||
|
||||
**只有 IP**
|
||||
Let's Encrypt 支持为公网 IP 签发证书(2026 年起可用,有效期约 7 天,须 `shortlived` profile)。用 [acme.sh](https://github.com/acmesh-official/acme.sh) 自动签发与续期,无需手动证书:
|
||||
|
||||
1. compose 里增加 acme-ip 服务(签发 + 每日检查自动续期):
|
||||
```yaml
|
||||
acme-ip:
|
||||
image: neilpang/acme.sh
|
||||
container_name: acme-ip
|
||||
command: daemon
|
||||
restart: always
|
||||
volumes:
|
||||
- certs:/acme.sh
|
||||
- html:/usr/share/nginx/html
|
||||
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||
networks: [proxy]
|
||||
```
|
||||
2. 首次签发(把 `<SERVER_IP>` 换成服务器公网 IP,IPv6 同样支持,多个 `-d` 可并列):
|
||||
```bash
|
||||
docker compose exec acme-ip acme.sh --issue --server letsencrypt \
|
||||
-d <SERVER_IP> --cert-profile shortlived --days 3 \
|
||||
--webroot /usr/share/nginx/html \
|
||||
--install-cert --cert-file /acme.sh/<SERVER_IP>.crt \
|
||||
--key-file /acme.sh/<SERVER_IP>.key \
|
||||
--reloadcmd "curl --unix-socket /var/run/docker.sock -X POST http://localhost/containers/nginx-proxy/kill?signal=HUP"
|
||||
```
|
||||
3. compose 里设 `VIRTUAL_HOST: '<SERVER_IP>'`、`WEBHOOK_URL: 'https://<SERVER_IP>/'`,无需 `WEBHOOK_CERT`。续期由 acme.sh daemon 自动完成(`--days 3` = 每 3 天续一次,证书 7 天有效有缓冲),续期成功后自动 HUP 通知 nginx-proxy 加载新证书。
|
||||
|
||||
限制:证书约 7 天有效;验证仅支持 http-01/tls-alpn-01(80 端口必须公网可达);不支持 DNS-01、私有 IP 与 IP 段;同一 IP 集合每 168 小时限签发 5 张。建议先用 `--server letsencrypt_test` 试签,成功后再切正式服务器。
|
||||
|
||||
Telegram 只接受 443/80/88/8443 端口。
|
||||
|
||||
<details>
|
||||
<summary>环境变量说明</summary>
|
||||
|
||||
| 变量 | 说明 |
|
||||
|---|---|
|
||||
| `TELOXIDE_TOKEN` | Bot token(必填) |
|
||||
| `PIXIV_REFRESH_TOKEN` | Pixiv 刷新令牌;未设置则禁用 Pixiv |
|
||||
| `BOT_ADMIN` | 管理员聊天 ID,逗号分隔;接收启动/停止通知 |
|
||||
| `EDIT_MESSAGE_TTL_SECONDS` | 转发前编辑记录过期秒数,默认 86400 |
|
||||
| `LINK_CACHE_TTL_SECONDS` | 链接结果缓存过期秒数,默认 604800(7 天) |
|
||||
| `RUST_LOG` | 日志级别 |
|
||||
| `LOCAL_USER_ID` | 容器内运行用户 UID,默认 9001 |
|
||||
| `VIRTUAL_HOST` | 对外域名或 IP,nginx-proxy 按此路由 |
|
||||
| `VIRTUAL_PORT` | bot 容器内监听端口,nginx-proxy 的转发目标 |
|
||||
| `ACME_HOST` | 域名部署:设为域名时由 acme-companion 自动签发/续期证书 |
|
||||
| `DEFAULT_HOST` | nginx-proxy 将未知 Host 的请求路由到该 vhost(IP 访问时需要) |
|
||||
| `DEFAULT_EMAIL` | acme-companion 证书通知邮箱 |
|
||||
| `WEBHOOK` | `true` 启用 webhook 模式(默认轮询) |
|
||||
| `WEBHOOK_LISTEN` / `WEBHOOK_PORT` | bot 容器内监听地址/端口 |
|
||||
| `WEBHOOK_URL` | 对外公网 HTTPS 地址(`https://域名/` 或 `https://IP/`) |
|
||||
| `WEBHOOK_SECRET_TOKEN` | 更新校验令牌(`X-Telegram-Bot-Api-Secret-Token`) |
|
||||
|
||||
</details>
|
||||
|
||||
## 命令
|
||||
|
||||
@@ -45,6 +112,6 @@ docker run --rm -d --name tgxmb --env-file .env -v ./data:/app/data tgxmb
|
||||
|
||||
## 备注
|
||||
|
||||
- 数据持久化于 `data/task_queue.db`,容器部署需挂载该目录
|
||||
- 数据持久化于 `data/task_queue.db`,compose 部署使用 bind mount `./data`(保持目录形式便于备份)
|
||||
- 运行环境需安装 ffmpeg(Docker 镜像已内置)
|
||||
- 测试:`cargo test --workspace`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "x-media"
|
||||
version = "1.0.2"
|
||||
version = "1.0.7"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
@@ -13,6 +13,7 @@ url = "2.5.2"
|
||||
bytes = "1"
|
||||
zip = "2"
|
||||
tempfile = "3"
|
||||
rand = "0.8"
|
||||
log = "0.4"
|
||||
tokio = { version = "1.40", features = ["time"] }
|
||||
|
||||
|
||||
@@ -14,6 +14,24 @@ impl Media {
|
||||
Media::Animated { thumbnail_url, .. } => Some(thumbnail_url),
|
||||
}
|
||||
}
|
||||
|
||||
/// A smaller variant of this media's file (used as the fallback when the
|
||||
/// primary URL or upload exceeds Telegram's size limits). None when no
|
||||
/// smaller variant exists (videos, animated gifs).
|
||||
pub fn smaller_url(&self) -> Option<&str> {
|
||||
match self {
|
||||
Media::Illustration {
|
||||
url,
|
||||
fallback_url,
|
||||
thumbnail_url,
|
||||
..
|
||||
} => fallback_url
|
||||
.as_deref()
|
||||
.or(thumbnail_url.as_deref())
|
||||
.filter(|smaller| *smaller != url),
|
||||
Media::Video { .. } | Media::Animated { .. } => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
|
||||
+131
-10
@@ -68,18 +68,73 @@ impl Fetched {
|
||||
/// caption.
|
||||
pub fn caption_with(&self, format: &str) -> String {
|
||||
match (&self.render_data, format.is_empty()) {
|
||||
(Some(data), false) => {
|
||||
let escaped = html_escape::encode_text(format).into_owned();
|
||||
escaped
|
||||
.replace("{url}", &data.url)
|
||||
.replace("{author}", &data.author)
|
||||
.replace("{author_url}", &data.author_url)
|
||||
.replace("{title}", &data.title)
|
||||
.replace("{tags}", &data.tags)
|
||||
}
|
||||
(Some(data), false) => caption_from_fields(
|
||||
format,
|
||||
"",
|
||||
&data.url,
|
||||
&data.author,
|
||||
&data.author_url,
|
||||
&data.title,
|
||||
&data.tags,
|
||||
),
|
||||
_ => self.caption.clone(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The pre-escaped placeholder values (author, author_url, title, tags)
|
||||
/// a caller needs to rebuild a caption later, e.g. for a cached post
|
||||
/// where the [`Fetched`] is no longer available.
|
||||
pub fn render_fields(&self) -> Option<(&str, &str, &str, &str)> {
|
||||
self.render_data.as_ref().map(|d| {
|
||||
(
|
||||
d.author.as_str(),
|
||||
d.author_url.as_str(),
|
||||
d.title.as_str(),
|
||||
d.tags.as_str(),
|
||||
)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Renders a user-supplied caption format from raw (already-escaped) field
|
||||
/// values with the same escaping/substitution rules as
|
||||
/// [`Fetched::caption_with`]. An empty format returns `built_in` unchanged.
|
||||
pub fn caption_from_fields(
|
||||
format: &str,
|
||||
built_in: &str,
|
||||
url: &str,
|
||||
author: &str,
|
||||
author_url: &str,
|
||||
title: &str,
|
||||
tags: &str,
|
||||
) -> String {
|
||||
if format.is_empty() {
|
||||
return built_in.to_string();
|
||||
}
|
||||
let escaped = html_escape::encode_text(format).into_owned();
|
||||
escaped
|
||||
.replace("{url}", url)
|
||||
.replace("{author}", author)
|
||||
.replace("{author_url}", author_url)
|
||||
.replace("{title}", title)
|
||||
.replace("{tags}", tags)
|
||||
}
|
||||
|
||||
/// Stable per-post cache key derived from any supported URL, so variant
|
||||
/// domains (x.com / twitter.com / fxtwitter.com, mobile, `/photo/N`
|
||||
/// suffixes) map to the same post. Returns `"twitter:<id>"`,
|
||||
/// `"pixiv:<id>"` or `"bsky:<handle>/<rkey>"`.
|
||||
pub fn cache_key(url: &str) -> Option<String> {
|
||||
if let Some(caps) = twitter::PATTERN.captures(url) {
|
||||
return Some(format!("twitter:{}", &caps[1]));
|
||||
}
|
||||
if let Some(caps) = pixiv::PATTERN.captures(url) {
|
||||
return Some(format!("pixiv:{}", &caps[1]));
|
||||
}
|
||||
if let Some(caps) = bsky::PATTERN.captures(url) {
|
||||
return Some(format!("bsky:{}/{}", &caps[1], &caps[2]));
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
@@ -89,6 +144,9 @@ pub enum FetchError {
|
||||
Pixiv(PixivError),
|
||||
NotFound,
|
||||
Blocked,
|
||||
/// The post exists but its content is withheld (twitter NSFW /
|
||||
/// age-restricted tweets come back as an empty `{}` from syndication).
|
||||
Sensitive,
|
||||
}
|
||||
|
||||
impl fmt::Display for FetchError {
|
||||
@@ -99,6 +157,7 @@ impl fmt::Display for FetchError {
|
||||
FetchError::Pixiv(e) => write!(f, "pixiv error: {e}"),
|
||||
FetchError::NotFound => write!(f, "not found"),
|
||||
FetchError::Blocked => write!(f, "blocked"),
|
||||
FetchError::Sensitive => write!(f, "content withheld (sensitive)"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -109,7 +168,7 @@ impl std::error::Error for FetchError {
|
||||
FetchError::Http(e) => Some(e),
|
||||
FetchError::Json(e) => Some(e),
|
||||
FetchError::Pixiv(e) => Some(e),
|
||||
FetchError::NotFound | FetchError::Blocked => None,
|
||||
FetchError::NotFound | FetchError::Blocked | FetchError::Sensitive => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -195,6 +254,19 @@ async fn fetch_once(url: &str) -> Result<Option<Fetched>, FetchError> {
|
||||
/// fetch of a media URL is blocked (hotlink protection), the bot downloads
|
||||
/// the file itself and uploads it via multipart. Site-appropriate headers:
|
||||
/// pixiv image hosts need the `Referer` header.
|
||||
/// Returns the Content-Length of a media URL, or `None` when the server does
|
||||
/// not report one. Used to check whether a file fits Telegram's size limits
|
||||
/// before downloading/uploading it.
|
||||
pub async fn media_size(url: &str) -> Result<Option<u64>, FetchError> {
|
||||
let mut request = CLIENT.get(url);
|
||||
let lower = url.to_ascii_lowercase();
|
||||
if lower.contains("pximg.net") {
|
||||
request = request.header("Referer", "https://www.pixiv.net/");
|
||||
}
|
||||
let response = request.send().await?;
|
||||
Ok(response.content_length())
|
||||
}
|
||||
|
||||
pub async fn download_media(url: &str) -> Result<bytes::Bytes, FetchError> {
|
||||
let mut request = CLIENT.get(url);
|
||||
let lower = url.to_ascii_lowercase();
|
||||
@@ -209,6 +281,55 @@ pub async fn download_media(url: &str) -> Result<bytes::Bytes, FetchError> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn cache_key_normalizes_domain_variants() {
|
||||
assert_eq!(
|
||||
cache_key("https://x.com/user/status/1234567890/photo/1"),
|
||||
Some("twitter:1234567890".into())
|
||||
);
|
||||
assert_eq!(
|
||||
cache_key("https://mobile.twitter.com/user/status/1234567890"),
|
||||
Some("twitter:1234567890".into())
|
||||
);
|
||||
assert_eq!(
|
||||
cache_key("https://fxtwitter.com/user/status/1234567890"),
|
||||
Some("twitter:1234567890".into())
|
||||
);
|
||||
assert_eq!(
|
||||
cache_key("https://www.pixiv.net/artworks/123456"),
|
||||
Some("pixiv:123456".into())
|
||||
);
|
||||
assert_eq!(
|
||||
cache_key("https://bsky.app/profile/handle.example/post/3lorem"),
|
||||
Some("bsky:handle.example/3lorem".into())
|
||||
);
|
||||
assert_eq!(cache_key("https://example.com/not-a-post"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn caption_from_fields_substitutes_and_escapes() {
|
||||
// The format string is escaped, the field values are substituted
|
||||
// verbatim (callers pass the already-escaped render data).
|
||||
let out = caption_from_fields(
|
||||
"see {author} at {url} — {title}",
|
||||
"",
|
||||
"https://x.com/u/status/1",
|
||||
"A & B",
|
||||
"https://x.com/u",
|
||||
"hello <world>",
|
||||
"",
|
||||
);
|
||||
assert_eq!(
|
||||
out,
|
||||
"see A & B at https://x.com/u/status/1 — hello <world>"
|
||||
);
|
||||
// Empty format keeps the built-in caption untouched.
|
||||
assert_eq!(
|
||||
caption_from_fields("", "built-in", "u", "a", "au", "t", "g"),
|
||||
"built-in"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn unsupported_url_returns_none() {
|
||||
let result = fetch("https://example.com/some/article").await;
|
||||
|
||||
@@ -0,0 +1,376 @@
|
||||
//! Authenticated fallback for tweets the public syndication endpoint refuses
|
||||
//! to serve (NSFW / age-restricted tweets come back as an empty `{}`).
|
||||
//!
|
||||
//! Mirrors nazurin's web API client ([`web.py`]) and is used *only* when
|
||||
//! syndication reports [`FetchError::Sensitive`]: the private GraphQL
|
||||
//! `TweetDetail` endpoint, authenticated with a browser session cookie from
|
||||
//! `TWITTER_AUTH_TOKEN` (the `auth_token` cookie value of a logged-in x.com
|
||||
//! session). A fresh random `ct0` is generated per call; X checks that the
|
||||
//! `x-csrf-token` header matches the cookie, not that it issued the value.
|
||||
//!
|
||||
//! [`web.py`]: https://github.com/y-young/nazurin/blob/master/nazurin/sites/twitter/api/web.py
|
||||
//!
|
||||
//! # Caveats
|
||||
//! - X rotates the GraphQL query id when it rolls the web app; if requests
|
||||
//! start failing, update [`TWEET_DETAIL_QUERY_ID`]. Fresh references from
|
||||
//! the actively maintained FxEmbed/FxEmbed: TweetDetail
|
||||
//! `R9IzzyzQBV87-DOWpcvDmw`, TweetResultByRestId `f2sagi1jweVHFkTUIHzmMQ`
|
||||
//! (the latter is anonymous and surfaces NSFW tweets as
|
||||
//! `reason: NsfwLoggedOut`).
|
||||
//! - `x-client-transaction-id` is only required for `SearchTimeline`
|
||||
//! (verified against FxEmbed's `proxy/allowlist.ts`) — TweetDetail works
|
||||
//! without it; no need for the nazurin home-page/JS-bundle derivation.
|
||||
|
||||
use std::sync::LazyLock;
|
||||
|
||||
use serde_json::{json, Value};
|
||||
|
||||
use crate::site::FetchError;
|
||||
|
||||
use super::interface::Tweet;
|
||||
|
||||
/// `auth_token` cookie of a logged-in x.com session; enables the fallback.
|
||||
/// Trimmed: a CRLF `.env` (Windows) leaves a trailing `\r` on the value,
|
||||
/// which would make the Cookie header invalid.
|
||||
static AUTH_TOKEN: LazyLock<Option<String>> = LazyLock::new(|| {
|
||||
std::env::var("TWITTER_AUTH_TOKEN")
|
||||
.ok()
|
||||
.map(|s| s.trim().to_string())
|
||||
.filter(|s| !s.is_empty())
|
||||
});
|
||||
|
||||
/// Public "logged in" client token used by the x.com web app.
|
||||
const LOGGED_IN_BEARER: &str =
|
||||
"Bearer AAAAAAAAAAAAAAAAAAAAANRILgAAAAAAnNwIzUejRCOuH5E6I8xnZz4puTs%3D1Zv7ttfk8LF81IUq16cHjhLTvJu4FA33AGWWjCpTnA";
|
||||
|
||||
/// `TweetDetail` query id (from nazurin; still valid as of 2026-08,
|
||||
/// corroborated by the current FxEmbed build — see module caveats).
|
||||
const TWEET_DETAIL_QUERY_ID: &str = "_8aYOgEDz35BrBcBal1-_w";
|
||||
|
||||
fn variables(id: &str) -> Value {
|
||||
json!({
|
||||
"focalTweetId": id,
|
||||
"with_rux_injections": false,
|
||||
"includePromotedContent": false,
|
||||
"withCommunity": true,
|
||||
"withQuickPromoteEligibilityTweetFields": false,
|
||||
"withBirdwatchNotes": false,
|
||||
"withVoice": true,
|
||||
})
|
||||
}
|
||||
|
||||
fn features() -> Value {
|
||||
json!({
|
||||
"rweb_video_screen_enabled": false,
|
||||
"profile_label_improvements_pcf_label_in_post_enabled": true,
|
||||
"rweb_tipjar_consumption_enabled": true,
|
||||
"verified_phone_label_enabled": false,
|
||||
"creator_subscriptions_tweet_preview_api_enabled": true,
|
||||
"responsive_web_graphql_timeline_navigation_enabled": true,
|
||||
"responsive_web_graphql_skip_user_profile_image_extensions_enabled": false,
|
||||
"premium_content_api_read_enabled": false,
|
||||
"communities_web_enable_tweet_community_results_fetch": true,
|
||||
"c9s_tweet_anatomy_moderator_badge_enabled": true,
|
||||
"responsive_web_grok_analyze_button_fetch_trends_enabled": false,
|
||||
"responsive_web_grok_analyze_post_followups_enabled": true,
|
||||
"responsive_web_jetfuel_frame": false,
|
||||
"responsive_web_grok_share_attachment_enabled": true,
|
||||
"articles_preview_enabled": true,
|
||||
"responsive_web_edit_tweet_api_enabled": true,
|
||||
"graphql_is_translatable_rweb_tweet_is_translatable_enabled": true,
|
||||
"view_counts_everywhere_api_enabled": true,
|
||||
"longform_notetweets_consumption_enabled": true,
|
||||
"responsive_web_twitter_article_tweet_consumption_enabled": true,
|
||||
"tweet_awards_web_tipping_enabled": false,
|
||||
"responsive_web_grok_show_grok_translated_post": false,
|
||||
"responsive_web_grok_analysis_button_from_backend": true,
|
||||
"creator_subscriptions_quote_tweet_preview_enabled": false,
|
||||
"freedom_of_speech_not_reach_fetch_enabled": true,
|
||||
"standardized_nudges_misinfo": true,
|
||||
"tweet_with_visibility_results_prefer_gql_limited_actions_policy_enabled": true,
|
||||
"longform_notetweets_rich_text_read_enabled": true,
|
||||
"longform_notetweets_inline_media_enabled": true,
|
||||
"responsive_web_grok_image_annotation_enabled": true,
|
||||
"responsive_web_enhance_cards_enabled": false,
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether the authenticated fallback is available.
|
||||
pub fn enabled() -> bool {
|
||||
AUTH_TOKEN.is_some()
|
||||
}
|
||||
|
||||
/// Fetches a tweet as the logged-in user via the private GraphQL API.
|
||||
/// Returns the syndication-shaped [`Tweet`] (media included for NSFW posts).
|
||||
pub async fn fetch(id: &str) -> Result<Tweet, FetchError> {
|
||||
let token = AUTH_TOKEN
|
||||
.as_deref()
|
||||
.ok_or(FetchError::Sensitive)?;
|
||||
// 16 random bytes as 32 hex chars: X rejects ct0 values of any other
|
||||
// length with 403 code 353 ("matching csrf cookie and header").
|
||||
let ct0: String = (0..16)
|
||||
.map(|_| format!("{:02x}", rand::random::<u8>()))
|
||||
.collect();
|
||||
|
||||
let response = crate::site::CLIENT
|
||||
.get(format!(
|
||||
"https://x.com/i/api/graphql/{TWEET_DETAIL_QUERY_ID}/TweetDetail"
|
||||
))
|
||||
.query(&[
|
||||
("variables", variables(id).to_string()),
|
||||
("features", features().to_string()),
|
||||
])
|
||||
.header("authorization", LOGGED_IN_BEARER)
|
||||
.header("x-csrf-token", &ct0)
|
||||
.header("x-twitter-auth-type", "OAuth2Session")
|
||||
.header("cookie", format!("auth_token={token}; ct0={ct0}"))
|
||||
.header("x-twitter-client-language", "en")
|
||||
.header("x-twitter-active-user", "yes")
|
||||
.header("referer", "https://x.com/")
|
||||
.send()
|
||||
.await?;
|
||||
if !response.status().is_success() {
|
||||
log::warn!("twitter auth fetch {id}: HTTP {}", response.status());
|
||||
return Err(FetchError::NotFound);
|
||||
}
|
||||
let text = response.text().await?;
|
||||
let json: Value = serde_json::from_str(&text)?;
|
||||
let result = parse_tweet_result(&json, id)?;
|
||||
let syndication_shape = to_syndication_shape(&result)
|
||||
.ok_or_else(|| FetchError::Json(serde_json::Error::io(std::io::Error::new(
|
||||
std::io::ErrorKind::InvalidData,
|
||||
"missing tweet fields in GraphQL response",
|
||||
))))?;
|
||||
Tweet::from_syndication_json(&syndication_shape.to_string()).map_err(FetchError::Json)
|
||||
}
|
||||
|
||||
/// Locates the tweet for `id` in a `TweetDetail` response and unwraps
|
||||
/// visibility wrappers / retweets, mirroring nazurin's `_process_response`.
|
||||
fn parse_tweet_result(json: &Value, id: &str) -> Result<Value, FetchError> {
|
||||
if let Some(errors) = json.get("errors").and_then(|e| e.as_array()) {
|
||||
let messages: Vec<&str> = errors
|
||||
.iter()
|
||||
.filter_map(|e| e.get("message").and_then(|m| m.as_str()))
|
||||
.collect();
|
||||
log::warn!("twitter auth fetch {id} failed: {}", messages.join("; "));
|
||||
return Err(FetchError::NotFound);
|
||||
}
|
||||
|
||||
let instructions = json
|
||||
.pointer("/data/threaded_conversation_with_injections_v2/instructions")
|
||||
.and_then(|v| v.as_array())
|
||||
.ok_or(FetchError::NotFound)?;
|
||||
for instruction in instructions {
|
||||
if instruction.get("type").and_then(|t| t.as_str()) != Some("TimelineAddEntries") {
|
||||
continue;
|
||||
}
|
||||
let entries = instruction
|
||||
.get("entries")
|
||||
.and_then(|e| e.as_array())
|
||||
.ok_or(FetchError::NotFound)?;
|
||||
let wanted = format!("tweet-{id}");
|
||||
for entry in entries {
|
||||
if entry.get("entryId").and_then(|i| i.as_str()) == Some(wanted.as_str()) {
|
||||
let result = entry
|
||||
.pointer("/content/itemContent/tweet_results/result")
|
||||
.ok_or(FetchError::NotFound)?;
|
||||
return normalize_tweet_result(result);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(FetchError::NotFound)
|
||||
}
|
||||
|
||||
/// Unwraps TweetTombstone/TweetUnavailable errors, the
|
||||
/// TweetWithVisibilityResults wrapper and retweets, returning the
|
||||
/// `{core, legacy, ...}` tweet object.
|
||||
fn normalize_tweet_result(result: &Value) -> Result<Value, FetchError> {
|
||||
match result.get("__typename").and_then(|t| t.as_str()) {
|
||||
Some("TweetTombstone") => {
|
||||
let text = result
|
||||
.pointer("/tombstone/text/text")
|
||||
.and_then(|t| t.as_str())
|
||||
.unwrap_or("tweet is unavailable");
|
||||
log::warn!("twitter auth fetch: tombstone: {text}");
|
||||
return Err(FetchError::NotFound);
|
||||
}
|
||||
Some("TweetUnavailable") => {
|
||||
let reason = result
|
||||
.get("reason")
|
||||
.and_then(|r| r.as_str())
|
||||
.unwrap_or("unknown");
|
||||
log::warn!("twitter auth fetch: tweet unavailable: {reason}");
|
||||
return Err(FetchError::NotFound);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
|
||||
// TweetWithVisibilityResults (e.g. limited replies) nests the real tweet.
|
||||
let tweet = result.get("tweet").unwrap_or(result);
|
||||
// A retweet's media lives on the original tweet.
|
||||
if let Some(original) = tweet.pointer("/legacy/retweeted_status_result/result") {
|
||||
return Ok(original.clone());
|
||||
}
|
||||
Ok(tweet.clone())
|
||||
}
|
||||
|
||||
/// Maps a GraphQL `{core, legacy, ...}` tweet onto the syndication JSON
|
||||
/// shape [`Tweet::from_syndication_json`] parses, so the existing text /
|
||||
/// media handling (t.co expansion, `name=orig`, mp4 variant) is reused.
|
||||
fn to_syndication_shape(tweet: &Value) -> Option<Value> {
|
||||
let legacy = tweet.get("legacy")?;
|
||||
let user = tweet.pointer("/core/user_results/result/legacy")?;
|
||||
Some(json!({
|
||||
"id_str": legacy.get("id_str"),
|
||||
"text": legacy.get("full_text"),
|
||||
"user": {
|
||||
"name": user.get("name"),
|
||||
"screen_name": user.get("screen_name"),
|
||||
},
|
||||
"possibly_sensitive": legacy.get("possibly_sensitive"),
|
||||
"display_text_range": legacy.get("display_text_range"),
|
||||
"entities": legacy.get("entities"),
|
||||
"mediaDetails": legacy.pointer("/extended_entities/media"),
|
||||
}))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn tweet_result() -> Value {
|
||||
json!({
|
||||
"__typename": "Tweet",
|
||||
"core": {
|
||||
"user_results": {
|
||||
"result": {
|
||||
"legacy": { "name": "Display Name", "screen_name": "nsfw_author" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"legacy": {
|
||||
"id_str": "2083868672721039569",
|
||||
"full_text": "nsfw content https://t.co/abc123",
|
||||
"display_text_range": [0, 12],
|
||||
"possibly_sensitive": true,
|
||||
"entities": {
|
||||
"urls": [
|
||||
{ "url": "https://t.co/abc123", "expanded_url": "https://example.com/x" }
|
||||
]
|
||||
},
|
||||
"extended_entities": {
|
||||
"media": [
|
||||
{
|
||||
"type": "photo",
|
||||
"media_url_https": "https://pbs.twimg.com/media/nsfw.jpg",
|
||||
"original_info": { "width": 1200, "height": 800 }
|
||||
},
|
||||
{
|
||||
"type": "video",
|
||||
"media_url_https": "https://pbs.twimg.com/thumb.jpg",
|
||||
"video_info": {
|
||||
"variants": [
|
||||
{ "content_type": "application/x-mpegURL", "url": "https://x.com/pl.m3u8" },
|
||||
{ "content_type": "video/mp4", "url": "https://video.twimg.com/nsfw.mp4" }
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn conversation(tweet: Value) -> Value {
|
||||
json!({
|
||||
"data": {
|
||||
"threaded_conversation_with_injections_v2": {
|
||||
"instructions": [
|
||||
{ "type": "TimelineAddEntries", "entries": [
|
||||
{ "entryId": "tweet-2083868672721039569",
|
||||
"content": { "itemContent": { "tweet_results": { "result": tweet } } } }
|
||||
]}
|
||||
]
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_graphql_tweet_into_fetched() {
|
||||
let json = conversation(tweet_result());
|
||||
let result = parse_tweet_result(&json, "2083868672721039569").unwrap();
|
||||
let shape = to_syndication_shape(&result).unwrap();
|
||||
let tweet = Tweet::from_syndication_json(&shape.to_string()).unwrap();
|
||||
let fetched: crate::site::Fetched = tweet.into();
|
||||
|
||||
assert!(fetched.sensitive);
|
||||
assert_eq!(fetched.media.len(), 2);
|
||||
match &fetched.media[0] {
|
||||
crate::media::Media::Illustration { url, .. } => {
|
||||
assert_eq!(url, "https://pbs.twimg.com/media/nsfw.jpg?name=orig");
|
||||
}
|
||||
other => panic!("expected illustration, got {other:?}"),
|
||||
}
|
||||
match &fetched.media[1] {
|
||||
crate::media::Media::Video { url, .. } => {
|
||||
assert_eq!(url, "https://video.twimg.com/nsfw.mp4");
|
||||
}
|
||||
other => panic!("expected video, got {other:?}"),
|
||||
}
|
||||
assert_eq!(fetched.source_url, "https://x.com/nsfw_author/status/2083868672721039569");
|
||||
// display_text_range cuts the trailing t.co link.
|
||||
assert_eq!(fetched.title, "nsfw content");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unwraps_retweet_to_original() {
|
||||
let original = tweet_result();
|
||||
let mut rt = tweet_result();
|
||||
rt["legacy"]["retweeted_status_result"] = json!({ "result": original });
|
||||
let json = conversation(rt);
|
||||
let result = parse_tweet_result(&json, "2083868672721039569").unwrap();
|
||||
assert!(result.pointer("/legacy/retweeted_status_result").is_none());
|
||||
assert_eq!(result.pointer("/legacy/id_str").unwrap(), "2083868672721039569");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn error_response_maps_to_not_found() {
|
||||
let json = json!({ "errors": [{ "message": "NsfwLoggedOut" }] });
|
||||
assert!(matches!(
|
||||
parse_tweet_result(&json, "1"),
|
||||
Err(FetchError::NotFound)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_entry_maps_to_not_found() {
|
||||
let json = conversation(json!({ "__typename": "Tweet" }));
|
||||
assert!(matches!(
|
||||
parse_tweet_result(&json, "999"),
|
||||
Err(FetchError::NotFound)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tombstone_maps_to_not_found() {
|
||||
let tombstone = json!({
|
||||
"__typename": "TweetTombstone",
|
||||
"tombstone": { "text": { "text": "Age-restricted adult content" } }
|
||||
});
|
||||
let json = conversation(tombstone);
|
||||
assert!(matches!(
|
||||
parse_tweet_result(&json, "2083868672721039569"),
|
||||
Err(FetchError::NotFound)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn visibility_wrapper_unwraps() {
|
||||
let inner = tweet_result();
|
||||
let wrapped = json!({ "__typename": "TweetWithVisibilityResults", "tweet": inner });
|
||||
let json = conversation(wrapped);
|
||||
let result = parse_tweet_result(&json, "2083868672721039569").unwrap();
|
||||
assert_eq!(result.get("__typename").unwrap(), "Tweet");
|
||||
}
|
||||
}
|
||||
@@ -19,7 +19,44 @@ pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> {
|
||||
.and_then(|caps| caps.get(1))
|
||||
.map(|m| m.as_str())
|
||||
.ok_or(FetchError::NotFound)?;
|
||||
Ok(fetch(id).await?.into())
|
||||
match fetch(id).await {
|
||||
Ok(tweet) => Ok(tweet.into()),
|
||||
// Syndication withholds NSFW/age-restricted tweets (empty `{}`).
|
||||
// Retry as the logged-in user when TWITTER_AUTH_TOKEN is set;
|
||||
// otherwise degrade to an empty result (the bot replies
|
||||
// "No media found").
|
||||
Err(FetchError::Sensitive) => {
|
||||
if super::auth::enabled() {
|
||||
match super::auth::fetch(id).await {
|
||||
Ok(tweet) => Ok(tweet.into()),
|
||||
Err(e) => {
|
||||
log::warn!("twitter auth fallback failed for {id}: {e}");
|
||||
Ok(empty_fetched(url))
|
||||
}
|
||||
}
|
||||
} else {
|
||||
log::info!(
|
||||
"tweet {id} is sensitive; set TWITTER_AUTH_TOKEN to fetch NSFW media"
|
||||
);
|
||||
Ok(empty_fetched(url))
|
||||
}
|
||||
}
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
}
|
||||
|
||||
/// A Fetched with no media for withheld tweets: the bot replies
|
||||
/// "No media found" and moves on instead of erroring.
|
||||
fn empty_fetched(url: &str) -> Fetched {
|
||||
Fetched {
|
||||
source_url: url.to_string(),
|
||||
caption: url.to_string(),
|
||||
title: String::new(),
|
||||
media: vec![],
|
||||
sensitive: true,
|
||||
render_data: None,
|
||||
_keep_alive: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Fetches a tweet from the syndication endpoint. Deleted/blocked tweets
|
||||
@@ -44,7 +81,16 @@ pub async fn fetch(id: &str) -> Result<Tweet, FetchError> {
|
||||
{
|
||||
return Err(FetchError::NotFound);
|
||||
}
|
||||
Ok(Tweet::from_syndication_json(&text).map_err(FetchError::Json)?)
|
||||
// NSFW / age-restricted tweets exist but are served as an empty `{}` —
|
||||
// they surface as FetchError::Sensitive so the caller can retry as a
|
||||
// logged-in user.
|
||||
if serde_json::from_str::<serde_json::Value>(&text)
|
||||
.map(|v| v.get("id_str").is_none())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
return Err(FetchError::Sensitive);
|
||||
}
|
||||
Tweet::from_syndication_json(&text).map_err(FetchError::Json)
|
||||
}
|
||||
|
||||
/// The syndication token: JS `((id / 1e15) * PI).toString(36)` (the
|
||||
@@ -129,7 +175,9 @@ impl Tweet {
|
||||
title: None,
|
||||
url: original_twimg_url(&item.media_url_https),
|
||||
thumbnail_url: None,
|
||||
fallback_url: None,
|
||||
// The param-less base URL is a reduced-size variant;
|
||||
// used as the fallback when the original is too large.
|
||||
fallback_url: Some(item.media_url_https.clone()),
|
||||
}),
|
||||
"video" => media.push(Media::Video {
|
||||
title: None,
|
||||
@@ -157,21 +205,17 @@ impl Tweet {
|
||||
}
|
||||
|
||||
/// The raw syndication `text` ends with the appended media short link
|
||||
/// (" https://t.co/wmI8McgXul"). `display_text_range` (UTF-16 indices) marks
|
||||
/// the visible text; a regex strips any remaining trailing t.co link when the
|
||||
/// range is absent or a tweet ends in a URL short link.
|
||||
/// (" https://t.co/wmI8McgXul"). `display_text_range` marks the visible text;
|
||||
/// a regex strips any remaining trailing t.co link when the range is absent
|
||||
/// or a tweet ends in a URL short link.
|
||||
///
|
||||
/// X reports these indices in Unicode **code points**, not UTF-16 units
|
||||
/// (verified against GraphQL responses containing emoji: cutting an emoji
|
||||
/// tweet by UTF-16 units silently drops the character after the emoji).
|
||||
fn strip_trailing_short_links(text: &str, display_text_range: Option<[usize; 2]>) -> String {
|
||||
let mut out = match display_text_range {
|
||||
Some([start, end]) if start < end => {
|
||||
let units: Vec<u16> = text
|
||||
.encode_utf16()
|
||||
.skip(start)
|
||||
.take(end - start)
|
||||
.collect();
|
||||
// Drop the replacement char that a surrogate cut at the boundary
|
||||
// would produce (the range end is a valid UTF-16 boundary in
|
||||
// practice, so this is just a safety net).
|
||||
String::from_utf16_lossy(&units).replace('\u{FFFD}', "")
|
||||
text.chars().skip(start).take(end - start).collect()
|
||||
}
|
||||
_ => text.to_string(),
|
||||
};
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
mod auth;
|
||||
mod interface;
|
||||
mod model;
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ pub struct SyndicationTweet {
|
||||
#[serde(default)]
|
||||
pub possibly_sensitive: Option<bool>,
|
||||
/// Visible-text span; the raw `text` field has the appended media short
|
||||
/// link after it. Indices are UTF-16 code units.
|
||||
/// link after it. Indices are Unicode code points (not UTF-16 units).
|
||||
#[serde(default, rename = "display_text_range")]
|
||||
pub display_text_range: Option<[usize; 2]>,
|
||||
#[serde(default)]
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "xmedia-bot"
|
||||
version = "1.0.2"
|
||||
version = "1.0.7"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
@@ -18,4 +18,8 @@ rusqlite = { version = "0.32", features = ["bundled"] }
|
||||
rand = "0.8"
|
||||
tempfile = "3"
|
||||
parking_lot = "0.12"
|
||||
png = "0.18"
|
||||
zune-jpeg = "0.5"
|
||||
fast_image_resize = "6"
|
||||
jpeg-encoder = "0.7"
|
||||
x-media = { path = "../x-media" }
|
||||
|
||||
@@ -10,6 +10,8 @@ pub struct Config {
|
||||
pub admin_ids: Vec<i64>,
|
||||
/// EDIT_MESSAGE_TTL_SECONDS, default 86400 (24h).
|
||||
pub edit_message_ttl: Duration,
|
||||
/// LINK_CACHE_TTL_SECONDS, default 604800 (7 days).
|
||||
pub link_cache_ttl: Duration,
|
||||
// Webhook settings (moved out of main; names/defaults unchanged).
|
||||
pub webhook_enabled: bool,
|
||||
pub webhook_url: Option<url::Url>,
|
||||
@@ -36,17 +38,30 @@ impl Config {
|
||||
.map(Duration::from_secs)
|
||||
.unwrap_or(Duration::from_secs(86400));
|
||||
|
||||
let link_cache_ttl = env::var("LINK_CACHE_TTL_SECONDS")
|
||||
.ok()
|
||||
.and_then(|s| s.parse::<u64>().ok())
|
||||
.map(Duration::from_secs)
|
||||
.unwrap_or(Duration::from_secs(7 * 24 * 3600));
|
||||
|
||||
let webhook_enabled = env::var("WEBHOOK")
|
||||
.is_ok_and(|v| matches!(v.to_lowercase().as_str(), "true" | "yes" | "1"));
|
||||
let webhook_url = env::var("WEBHOOK_URL").ok().and_then(|s| s.parse().ok());
|
||||
let webhook_listen = env::var("WEBHOOK_LISTEN").ok().and_then(|s| s.parse().ok());
|
||||
let webhook_port = env::var("WEBHOOK_PORT").ok().and_then(|s| s.parse().ok());
|
||||
let webhook_cert = env::var("WEBHOOK_CERT").ok();
|
||||
let webhook_secret_token = env::var("WEBHOOK_SECRET_TOKEN").ok();
|
||||
// Empty strings count as unset (e.g. `-e WEBHOOK_CERT=` to disable a
|
||||
// value that would otherwise come from `.env`).
|
||||
let webhook_cert = env::var("WEBHOOK_CERT")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty());
|
||||
let webhook_secret_token = env::var("WEBHOOK_SECRET_TOKEN")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty());
|
||||
|
||||
Config {
|
||||
admin_ids,
|
||||
edit_message_ttl,
|
||||
link_cache_ttl,
|
||||
webhook_enabled,
|
||||
webhook_url,
|
||||
webhook_listen,
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
use crate::config::Config;
|
||||
use crate::link_cache::{CachedMediaKind, CachedPost, LinkCache};
|
||||
use crate::queue::PersistentTaskQueue;
|
||||
use crate::send::{self, MediaItemPayload, Task};
|
||||
use crate::state::{ChatStore, unix_now};
|
||||
use crate::state::{ChatData, ChatStore, unix_now};
|
||||
use std::collections::HashSet;
|
||||
use std::sync::LazyLock;
|
||||
use teloxide::prelude::*;
|
||||
use tokio::sync::Semaphore;
|
||||
use teloxide::types::{
|
||||
CallbackQuery, ChatAction, ChatId, ChatKind, InlineQuery, InlineQueryResult,
|
||||
InlineQueryResultMpeg4Gif, InlineQueryResultPhoto, InlineQueryResultVideo, Message,
|
||||
@@ -19,8 +21,18 @@ pub static CHAT_STORE: LazyLock<ChatStore> = LazyLock::new(|| {
|
||||
});
|
||||
pub static TASK_QUEUE: LazyLock<PersistentTaskQueue> =
|
||||
LazyLock::new(|| PersistentTaskQueue::new("data/task_queue.db"));
|
||||
pub static LINK_CACHE: LazyLock<LinkCache> =
|
||||
LazyLock::new(|| LinkCache::open("data/task_queue.db"));
|
||||
pub static CONFIG: LazyLock<Config> = LazyLock::new(Config::load);
|
||||
|
||||
/// Cap on concurrent per-URL processing. teloxide dispatches updates to a
|
||||
/// per-chat worker that handles them sequentially, so a batch-forward of many
|
||||
/// messages would otherwise be processed one at a time (fetch + send each,
|
||||
/// roughly a second per message). Moving the work into spawned tasks trades
|
||||
/// per-chat reply ordering for throughput; the semaphore bounds how many run
|
||||
/// at once so a big burst cannot hammer Telegram's rate limits.
|
||||
static URL_TASKS: LazyLock<Semaphore> = LazyLock::new(|| Semaphore::new(8));
|
||||
|
||||
#[derive(BotCommands, Clone)]
|
||||
#[command(rename_rule = "snake_case", description = "")]
|
||||
enum Command {
|
||||
@@ -322,22 +334,29 @@ fn thumbnail_for(media: &Media) -> Option<String> {
|
||||
}
|
||||
|
||||
fn media_to_payload(media: &Media, sensitive: bool) -> MediaItemPayload {
|
||||
let fallback_url = media.smaller_url().map(str::to_string);
|
||||
match media {
|
||||
// A gif inside a group becomes a video item; a lone gif takes the
|
||||
// animation path (see url_media).
|
||||
Media::Illustration { .. } => MediaItemPayload::Photo {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
Media::Video { .. } => MediaItemPayload::Video {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
thumbnail: thumbnail_for(media),
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
Media::Animated { .. } => MediaItemPayload::Video {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
thumbnail: thumbnail_for(media),
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -350,11 +369,148 @@ async fn enqueue_retry(task: Task, delay_seconds: f64) {
|
||||
}
|
||||
}
|
||||
|
||||
/// Sends a task and handles the outcome: post-send actions on success, retry
|
||||
/// enqueue on retryable failure, reply + link-cache invalidation on
|
||||
/// permanent failure (a stale cached file id must not repeat forever).
|
||||
async fn dispatch_send(bot: Bot, message: &Message, task: &Task, url: &str) {
|
||||
let result = match task {
|
||||
Task::SendAnimation { .. } => send::send_animation(&bot, task).await,
|
||||
Task::SendMediaSequence { .. } => send::send_media_sequence(&bot, task).await,
|
||||
Task::ForwardMessages { .. } => unreachable!(),
|
||||
};
|
||||
match result {
|
||||
Ok(message_ids) => {
|
||||
log::info!("sent {} message(s) for {url}", message_ids.len());
|
||||
send::post_send_actions(&bot, task, message_ids).await;
|
||||
}
|
||||
Err(send::SendError::Retryable { delay_seconds, task }) => {
|
||||
log::info!("send for {url} failed, queued for retry in {delay_seconds:.1}s");
|
||||
enqueue_retry(task, delay_seconds).await;
|
||||
let _ = reply(bot, message.clone(), "Send failed. Task queued for retry.").await;
|
||||
}
|
||||
Err(send::SendError::Permanent {
|
||||
message: err_message,
|
||||
task,
|
||||
}) => {
|
||||
send::invalidate_cache(&task).await;
|
||||
log::error!("send for {url} failed permanently: {err_message}");
|
||||
let _ = reply(bot, message.clone(), format!("Send failed: {err_message}")).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Builds the send task from ready-made items, sharing the payload shape
|
||||
/// between the fresh-fetch and link-cache paths.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn build_send_task(
|
||||
chat_data: &ChatData,
|
||||
message: &Message,
|
||||
source_url: String,
|
||||
caption: String,
|
||||
items: Vec<MediaItemPayload>,
|
||||
cache_data: Option<CachedPost>,
|
||||
) -> Task {
|
||||
let chat_id = message.chat.id.0;
|
||||
if items.len() == 1 && matches!(items[0], MediaItemPayload::Animation { .. }) {
|
||||
Task::SendAnimation {
|
||||
chat_id,
|
||||
reply_to_message_id: message.id.0 as i64,
|
||||
caption,
|
||||
animation: items.into_iter().next().unwrap(),
|
||||
source_url,
|
||||
edit_before_forward: chat_data.edit_before_forward,
|
||||
forward_channel_id: chat_data.forward_channel_id,
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(message.id.0 as i64),
|
||||
cache_data,
|
||||
}
|
||||
} else {
|
||||
Task::SendMediaSequence {
|
||||
chat_id,
|
||||
reply_to_message_id: message.id.0 as i64,
|
||||
caption,
|
||||
media_batches: send::chunk_media_items(items),
|
||||
batch_index: 0,
|
||||
sent_message_ids: vec![],
|
||||
source_url,
|
||||
edit_before_forward: chat_data.edit_before_forward,
|
||||
forward_channel_id: chat_data.forward_channel_id,
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(message.id.0 as i64),
|
||||
cache_data,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn url_media(bot: Bot, message: &Message, url: &str) {
|
||||
let chat_id = message.chat.id.0;
|
||||
if let Err(e) = bot.send_chat_action(ChatId(chat_id), ChatAction::Typing).await {
|
||||
log::error!("send_chat_action failed: {e}");
|
||||
}
|
||||
|
||||
// Link cache: a post sent before is re-sent from Telegram file ids —
|
||||
// no source-site request, no download, no upload. Keyed by the
|
||||
// normalized post id so x.com / fxtwitter / /photo/N variants collide.
|
||||
if let Some(key) = x_media::site::cache_key(url)
|
||||
&& let Some(cached) = LINK_CACHE.get(&key, CONFIG.link_cache_ttl).await
|
||||
{
|
||||
log::info!("link cache hit for {url}");
|
||||
let chat_data = CHAT_STORE.get(chat_id).await;
|
||||
let site = key.split(':').next().unwrap_or("unknown");
|
||||
let format = chat_data
|
||||
.message_format
|
||||
.get(site)
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let caption = if format.is_empty() {
|
||||
cached.caption.clone()
|
||||
} else {
|
||||
x_media::site::caption_from_fields(
|
||||
&format,
|
||||
"",
|
||||
&cached.url,
|
||||
&cached.author,
|
||||
&cached.author_url,
|
||||
&cached.title,
|
||||
&cached.tags,
|
||||
)
|
||||
};
|
||||
let items: Vec<MediaItemPayload> = cached
|
||||
.media
|
||||
.iter()
|
||||
.map(|m| match m.kind {
|
||||
CachedMediaKind::Photo => MediaItemPayload::Photo {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
fallback_url: None,
|
||||
file_id: true,
|
||||
},
|
||||
CachedMediaKind::Video => MediaItemPayload::Video {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
thumbnail: None,
|
||||
fallback_url: None,
|
||||
file_id: true,
|
||||
},
|
||||
CachedMediaKind::Animation => MediaItemPayload::Animation {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
file_id: true,
|
||||
},
|
||||
})
|
||||
.collect();
|
||||
let task = build_send_task(
|
||||
&chat_data,
|
||||
message,
|
||||
cached.url.clone(),
|
||||
caption,
|
||||
items,
|
||||
Some(cached),
|
||||
);
|
||||
dispatch_send(bot, message, &task, url).await;
|
||||
return;
|
||||
}
|
||||
|
||||
log::info!("fetching {url}");
|
||||
match x_media::site::fetch(url).await {
|
||||
// Unsupported links are ignored silently (Python parity).
|
||||
@@ -384,66 +540,34 @@ async fn url_media(bot: Bot, message: &Message, url: &str) {
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let caption = fetched.caption_with(&format);
|
||||
let task = if fetched.media.len() == 1
|
||||
&& matches!(fetched.media[0], Media::Animated { .. })
|
||||
{
|
||||
Task::SendAnimation {
|
||||
chat_id,
|
||||
reply_to_message_id: message.id.0 as i64,
|
||||
caption: caption.clone(),
|
||||
animation: MediaItemPayload::Animation {
|
||||
media: fetched.media[0].url().to_string(),
|
||||
has_spoiler: fetched.sensitive,
|
||||
},
|
||||
source_url: fetched.source_url.clone(),
|
||||
edit_before_forward: chat_data.edit_before_forward,
|
||||
forward_channel_id: chat_data.forward_channel_id,
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(message.id.0 as i64),
|
||||
// Raw render data for the link cache; the send fills in the
|
||||
// Telegram file ids and persists the entry.
|
||||
let cache_data = fetched.render_fields().map(|(author, author_url, title, tags)| {
|
||||
CachedPost {
|
||||
url: fetched.source_url.clone(),
|
||||
caption: fetched.caption.clone(),
|
||||
title: title.to_string(),
|
||||
author: author.to_string(),
|
||||
author_url: author_url.to_string(),
|
||||
tags: tags.to_string(),
|
||||
sensitive: fetched.sensitive,
|
||||
media: vec![],
|
||||
}
|
||||
} else {
|
||||
let items: Vec<MediaItemPayload> = fetched
|
||||
.media
|
||||
.iter()
|
||||
.map(|media| media_to_payload(media, fetched.sensitive))
|
||||
.collect();
|
||||
Task::SendMediaSequence {
|
||||
chat_id,
|
||||
reply_to_message_id: message.id.0 as i64,
|
||||
caption: caption.clone(),
|
||||
media_batches: send::chunk_media_items(items),
|
||||
batch_index: 0,
|
||||
sent_message_ids: vec![],
|
||||
source_url: fetched.source_url.clone(),
|
||||
edit_before_forward: chat_data.edit_before_forward,
|
||||
forward_channel_id: chat_data.forward_channel_id,
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(message.id.0 as i64),
|
||||
}
|
||||
};
|
||||
let result = match &task {
|
||||
Task::SendAnimation { .. } => send::send_animation(&bot, &task).await,
|
||||
Task::SendMediaSequence { .. } => send::send_media_sequence(&bot, &task).await,
|
||||
Task::ForwardMessages { .. } => unreachable!(),
|
||||
};
|
||||
match result {
|
||||
Ok(message_ids) => {
|
||||
log::info!("sent {} message(s) for {url}", message_ids.len());
|
||||
send::post_send_actions(&bot, &task, message_ids).await;
|
||||
}
|
||||
Err(send::SendError::Retryable { delay_seconds, task }) => {
|
||||
log::info!("send for {url} failed, queued for retry in {delay_seconds:.1}s");
|
||||
enqueue_retry(task, delay_seconds).await;
|
||||
let _ = reply(bot, message.clone(), "Send failed. Task queued for retry.").await;
|
||||
}
|
||||
Err(send::SendError::Permanent {
|
||||
message: err_message,
|
||||
..
|
||||
}) => {
|
||||
log::error!("send for {url} failed permanently: {err_message}");
|
||||
let _ = reply(bot, message.clone(), format!("Send failed: {err_message}")).await;
|
||||
}
|
||||
}
|
||||
});
|
||||
let items: Vec<MediaItemPayload> = fetched
|
||||
.media
|
||||
.iter()
|
||||
.map(|media| media_to_payload(media, fetched.sensitive))
|
||||
.collect();
|
||||
let task = build_send_task(
|
||||
&chat_data,
|
||||
message,
|
||||
fetched.source_url.clone(),
|
||||
caption,
|
||||
items,
|
||||
cache_data,
|
||||
);
|
||||
dispatch_send(bot, message, &task, url).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -477,7 +601,13 @@ pub async fn message_handler(bot: Bot, message: Message) -> Result<(), RequestEr
|
||||
log::info!("extracted {} URL(s): {urls:?}", urls.len());
|
||||
}
|
||||
for url in urls {
|
||||
url_media(bot.clone(), &message, &url).await;
|
||||
let bot = bot.clone();
|
||||
let message = message.clone();
|
||||
tokio::spawn(async move {
|
||||
// Held for the whole task; the semaphore is never closed.
|
||||
let _permit = URL_TASKS.acquire().await.expect("URL semaphore closed");
|
||||
url_media(bot, &message, &url).await;
|
||||
});
|
||||
}
|
||||
}
|
||||
respond(())
|
||||
@@ -502,11 +632,19 @@ pub async fn inline_query_handler(bot: Bot, query: InlineQuery) -> Result<(), Re
|
||||
.unwrap_or_else(|| url.clone());
|
||||
let caption = fetched.caption.clone();
|
||||
let result = match media {
|
||||
Media::Illustration { .. } => InlineQueryResult::Photo(
|
||||
InlineQueryResultPhoto::new(id, url, thumbnail)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html),
|
||||
),
|
||||
Media::Illustration { .. } => {
|
||||
// Inline photo results have their own (smaller) size
|
||||
// cap; use the reduced variant when one exists.
|
||||
let photo_url = media
|
||||
.smaller_url()
|
||||
.and_then(|u| url::Url::parse(u).ok())
|
||||
.unwrap_or_else(|| url.clone());
|
||||
InlineQueryResult::Photo(
|
||||
InlineQueryResultPhoto::new(id, photo_url, thumbnail)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html),
|
||||
)
|
||||
}
|
||||
Media::Video { .. } => InlineQueryResult::Video(
|
||||
InlineQueryResultVideo::new(
|
||||
id,
|
||||
|
||||
@@ -0,0 +1,230 @@
|
||||
//! Persistent cache of successfully sent posts.
|
||||
//!
|
||||
//! After a media send succeeds, the raw render data plus the Telegram
|
||||
//! `file_id`s of the sent items are stored keyed by [`crate::site` cache
|
||||
//! key]. A repeated link is then answered entirely from local state — no
|
||||
//! re-fetch of the source site, no re-upload — and no media file is stored
|
||||
//! on disk (the file ids point at Telegram's servers). Entries expire after
|
||||
//! [`Config::link_cache_ttl`]; a stale entry is dropped lazily on read and
|
||||
//! by the periodic prune in `main`.
|
||||
|
||||
use rusqlite::{params, Connection};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::time::Duration;
|
||||
|
||||
#[derive(Serialize, Deserialize, Clone, Debug, PartialEq)]
|
||||
#[serde(rename_all = "snake_case")]
|
||||
pub enum CachedMediaKind {
|
||||
Photo,
|
||||
Video,
|
||||
Animation,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize, Clone, Debug)]
|
||||
pub struct CachedMedia {
|
||||
pub kind: CachedMediaKind,
|
||||
pub file_id: String,
|
||||
}
|
||||
|
||||
/// Everything needed to re-send a post without touching the source site:
|
||||
/// the canonical URL, pre-escaped caption fields, and the file ids produced
|
||||
/// by the original successful send.
|
||||
#[derive(Serialize, Deserialize, Clone, Debug)]
|
||||
pub struct CachedPost {
|
||||
pub url: String,
|
||||
/// The site's built-in caption (used when the chat has no format
|
||||
/// override).
|
||||
pub caption: String,
|
||||
pub title: String,
|
||||
pub author: String,
|
||||
pub author_url: String,
|
||||
pub tags: String,
|
||||
pub sensitive: bool,
|
||||
pub media: Vec<CachedMedia>,
|
||||
}
|
||||
|
||||
/// SQLite-backed cache sharing `data/task_queue.db` with the queue and chat
|
||||
/// state (same `open_db` pattern: busy timeout, `spawn_blocking` I/O).
|
||||
pub struct LinkCache {
|
||||
db_path: String,
|
||||
}
|
||||
|
||||
fn open_db(path: &str) -> rusqlite::Result<Connection> {
|
||||
let conn = Connection::open(path)?;
|
||||
conn.busy_timeout(Duration::from_secs(5))?;
|
||||
Ok(conn)
|
||||
}
|
||||
|
||||
impl LinkCache {
|
||||
pub fn open(db_path: &str) -> Self {
|
||||
if let Ok(conn) = Connection::open(db_path)
|
||||
&& let Err(e) = conn.execute_batch(
|
||||
"CREATE TABLE IF NOT EXISTS link_cache (url TEXT PRIMARY KEY, \
|
||||
payload TEXT NOT NULL, created_at REAL NOT NULL);",
|
||||
)
|
||||
{
|
||||
log::error!("failed to initialize link cache schema: {e}");
|
||||
}
|
||||
Self {
|
||||
db_path: db_path.to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns the cached post if present and not expired; a stale entry is
|
||||
/// removed on the spot.
|
||||
pub async fn get(&self, key: &str, ttl: Duration) -> Option<CachedPost> {
|
||||
let db_path = self.db_path.clone();
|
||||
let key = key.to_string();
|
||||
let ttl = ttl.as_secs_f64();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<Option<CachedPost>> {
|
||||
let conn = open_db(&db_path)?;
|
||||
let mut stmt =
|
||||
conn.prepare("SELECT payload, created_at FROM link_cache WHERE url = ?1")?;
|
||||
let mut rows = stmt.query(params![key])?;
|
||||
let Some(row) = rows.next()? else {
|
||||
return Ok(None);
|
||||
};
|
||||
let payload: String = row.get(0)?;
|
||||
let created_at: f64 = row.get(1)?;
|
||||
if now_f64() - created_at > ttl {
|
||||
conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key])?;
|
||||
return Ok(None);
|
||||
}
|
||||
serde_json::from_str(&payload).map(Some).map_err(|e| {
|
||||
rusqlite::Error::ToSqlConversionFailure(Box::new(e))
|
||||
})
|
||||
})
|
||||
.await
|
||||
.expect("link cache read worker panicked")
|
||||
.unwrap_or_else(|e| {
|
||||
log::error!("link cache read failed: {e}");
|
||||
None
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn put(&self, key: &str, post: &CachedPost) {
|
||||
let db_path = self.db_path.clone();
|
||||
let key = key.to_string();
|
||||
let payload = serde_json::to_string(post).expect("cached post serializes");
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<()> {
|
||||
let conn = open_db(&db_path)?;
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO link_cache (url, payload, created_at) VALUES (?1, ?2, ?3)",
|
||||
params![key, payload, now_f64()],
|
||||
)?;
|
||||
Ok(())
|
||||
})
|
||||
.await
|
||||
.expect("link cache write worker panicked")
|
||||
.unwrap_or_else(|e| log::error!("link cache write failed: {e}"));
|
||||
}
|
||||
|
||||
/// Drops an entry (e.g. a cached file id that turned out invalid).
|
||||
pub async fn remove(&self, key: &str) {
|
||||
let db_path = self.db_path.clone();
|
||||
let key = key.to_string();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<()> {
|
||||
let conn = open_db(&db_path)?;
|
||||
conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key])?;
|
||||
Ok(())
|
||||
})
|
||||
.await
|
||||
.expect("link cache delete worker panicked")
|
||||
.unwrap_or_else(|e| log::error!("link cache delete failed: {e}"));
|
||||
}
|
||||
|
||||
/// Removes expired entries; returns how many were deleted.
|
||||
pub async fn prune(&self, ttl: Duration) -> usize {
|
||||
let db_path = self.db_path.clone();
|
||||
let cutoff = now_f64() - ttl.as_secs_f64();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<usize> {
|
||||
let conn = open_db(&db_path)?;
|
||||
conn.execute(
|
||||
"DELETE FROM link_cache WHERE created_at < ?1",
|
||||
params![cutoff],
|
||||
)
|
||||
})
|
||||
.await
|
||||
.expect("link cache prune worker panicked")
|
||||
.unwrap_or_else(|e| {
|
||||
log::error!("link cache prune failed: {e}");
|
||||
0
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn now_f64() -> f64 {
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_secs_f64())
|
||||
.unwrap_or(0.0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn entry() -> CachedPost {
|
||||
CachedPost {
|
||||
url: "https://x.com/u/status/1".into(),
|
||||
caption: "cap".into(),
|
||||
title: "t".into(),
|
||||
author: "a".into(),
|
||||
author_url: "au".into(),
|
||||
tags: "".into(),
|
||||
sensitive: true,
|
||||
media: vec![CachedMedia {
|
||||
kind: CachedMediaKind::Photo,
|
||||
file_id: "AgAC...".into(),
|
||||
}],
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_get_roundtrip() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let cache = LinkCache::open(dir.path().join("c.db").to_str().unwrap());
|
||||
cache.put("twitter:1", &entry()).await;
|
||||
let got = cache.get("twitter:1", Duration::from_secs(3600)).await;
|
||||
assert!(got.is_some());
|
||||
let got = got.unwrap();
|
||||
assert_eq!(got.url, "https://x.com/u/status/1");
|
||||
assert_eq!(got.media[0].file_id, "AgAC...");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn expired_entry_removed_on_read() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let cache = LinkCache::open(dir.path().join("c.db").to_str().unwrap());
|
||||
cache.put("twitter:1", &entry()).await;
|
||||
// Force the row into the past so a 1s TTL expires it.
|
||||
{
|
||||
let conn = Connection::open(dir.path().join("c.db")).unwrap();
|
||||
conn.execute(
|
||||
"UPDATE link_cache SET created_at = created_at - 100",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
assert!(cache.get("twitter:1", Duration::from_secs(1)).await.is_none());
|
||||
assert!(cache.get("twitter:1", Duration::from_secs(3600)).await.is_none());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn remove_and_prune() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let cache = LinkCache::open(dir.path().join("c.db").to_str().unwrap());
|
||||
cache.put("twitter:1", &entry()).await;
|
||||
cache.put("pixiv:2", &entry()).await;
|
||||
cache.remove("twitter:1").await;
|
||||
assert!(cache.get("twitter:1", Duration::from_secs(3600)).await.is_none());
|
||||
assert!(cache.get("pixiv:2", Duration::from_secs(3600)).await.is_some());
|
||||
{
|
||||
let conn = Connection::open(dir.path().join("c.db")).unwrap();
|
||||
conn.execute("UPDATE link_cache SET created_at = created_at - 100", [])
|
||||
.unwrap();
|
||||
}
|
||||
assert_eq!(cache.prune(Duration::from_secs(1)).await, 1);
|
||||
assert!(cache.get("pixiv:2", Duration::from_secs(3600)).await.is_none());
|
||||
}
|
||||
}
|
||||
@@ -1,18 +1,39 @@
|
||||
use dotenv::dotenv;
|
||||
use teloxide::dptree::endpoint;
|
||||
use teloxide::stop::StopToken;
|
||||
use teloxide::types::{ChatId, InputFile, MessageId};
|
||||
use teloxide::update_listeners::webhooks;
|
||||
use teloxide::update_listeners::{self, webhooks, UpdateListener};
|
||||
use teloxide::prelude::*;
|
||||
use tokio::sync::watch;
|
||||
use x_media::site;
|
||||
|
||||
mod config;
|
||||
mod handlers;
|
||||
mod link_cache;
|
||||
mod photo;
|
||||
mod queue;
|
||||
mod send;
|
||||
mod state;
|
||||
|
||||
use handlers::{CHAT_STORE, CONFIG, TASK_QUEUE};
|
||||
use handlers::{CHAT_STORE, CONFIG, LINK_CACHE, TASK_QUEUE};
|
||||
|
||||
/// Docker `stop` / `compose down` delivers SIGTERM, which teloxide's ctrlc
|
||||
/// handler (SIGINT only) never sees — without this the process would die
|
||||
/// before the graceful shutdown below (admin notice, queue drain). Stopping
|
||||
/// the token unwinds the dispatcher exactly like Ctrl+C does.
|
||||
#[cfg(unix)]
|
||||
fn spawn_sigterm_handler(stop_token: StopToken) {
|
||||
tokio::spawn(async move {
|
||||
let mut sigterm = tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate())
|
||||
.expect("failed to install SIGTERM handler");
|
||||
sigterm.recv().await;
|
||||
log::info!("SIGTERM received, stopping the dispatcher");
|
||||
stop_token.stop();
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(not(unix))]
|
||||
fn spawn_sigterm_handler(_stop_token: StopToken) {}
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() {
|
||||
@@ -66,6 +87,10 @@ async fn main() {
|
||||
}
|
||||
let ttl = CONFIG.edit_message_ttl;
|
||||
let removed = CHAT_STORE.prune_expired(ttl).await;
|
||||
let pruned = LINK_CACHE.prune(CONFIG.link_cache_ttl).await;
|
||||
if pruned > 0 {
|
||||
log::info!("link cache: pruned {pruned} expired entr(ies)");
|
||||
}
|
||||
for (chat_id, prompt_message_id) in removed {
|
||||
// If the prompt was already deleted, this fails with a
|
||||
// 400 "message to edit not found" — log and ignore.
|
||||
@@ -96,7 +121,8 @@ async fn main() {
|
||||
.webhook_url
|
||||
.clone()
|
||||
.expect("WEBHOOK_URL is not set");
|
||||
bot.set_webhook(url.clone()).await.unwrap();
|
||||
// `webhooks::axum` calls set_webhook itself (with the full options,
|
||||
// secret token included) — no explicit registration here.
|
||||
let listen = CONFIG.webhook_listen.expect("WEBHOOK_LISTEN is not set");
|
||||
let port = CONFIG.webhook_port.expect("WEBHOOK_PORT is not set");
|
||||
let mut options = webhooks::Options::new((listen, port).into(), url);
|
||||
@@ -107,20 +133,36 @@ async fn main() {
|
||||
options = options.secret_token(secret.clone());
|
||||
}
|
||||
|
||||
let mut listener = webhooks::axum(bot.clone(), options)
|
||||
.await
|
||||
.expect("Failed to create webhook listener");
|
||||
let stop_token = listener.stop_token();
|
||||
spawn_sigterm_handler(stop_token);
|
||||
|
||||
dispatcher
|
||||
.dispatch_with_listener(
|
||||
webhooks::axum(bot.clone(), options)
|
||||
.await
|
||||
.expect("Failed to create webhook listener"),
|
||||
listener,
|
||||
LoggingErrorHandler::with_custom_text("Error from update listener"),
|
||||
)
|
||||
.await;
|
||||
} else {
|
||||
log::info!("running in polling mode");
|
||||
dispatcher.dispatch().await;
|
||||
// Same listener `dispatch()` builds internally — using
|
||||
// `dispatch_with_listener` just exposes its stop token so SIGTERM can
|
||||
// unwind the dispatcher before the graceful shutdown below.
|
||||
let mut listener = update_listeners::polling_default(bot.clone()).await;
|
||||
let stop_token = listener.stop_token();
|
||||
spawn_sigterm_handler(stop_token);
|
||||
|
||||
dispatcher
|
||||
.dispatch_with_listener(
|
||||
listener,
|
||||
LoggingErrorHandler::with_custom_text("Error from update listener"),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
// Graceful stop (Ctrl+C): stop the sweep, notify the admin, drain the queue.
|
||||
// Graceful stop (Ctrl+C / SIGTERM): stop the sweep, notify the admin, drain the queue.
|
||||
log::info!("Stopping bot");
|
||||
let _ = stop_tx.send(true);
|
||||
if let Some(admin) = CONFIG.admin_ids.first() {
|
||||
|
||||
@@ -0,0 +1,515 @@
|
||||
//! Pure-Rust photo processing: brings a downloaded photo within Telegram's
|
||||
//! limits (width + height ≤ 10000 px, bytes ≤ 10 MiB) without ffmpeg.
|
||||
//!
|
||||
//! Stack: `png` (image-png) for PNG decode/encode, `zune-jpeg` for JPEG
|
||||
//! decode, `fast_image_resize` (Lanczos3) for downsampling, `jpeg-encoder`
|
||||
//! for JPEG output.
|
||||
//!
|
||||
//! Bit-depth rule: a PNG above 24 bits (32-bit RGBA or 16-bit per channel)
|
||||
//! is reduced to 24-bit RGB; 24-bit and lower depths are left untouched —
|
||||
//! gray stays gray, never upconverted. The only upconversion is palette
|
||||
//! expansion, which resampling requires. Alpha is flattened onto white (JPEG
|
||||
//! and 24-bit RGB have no alpha channel).
|
||||
|
||||
use std::io::Write;
|
||||
|
||||
use fast_image_resize as fir;
|
||||
use tempfile::NamedTempFile;
|
||||
|
||||
/// Telegram rejects photos whose width + height exceed this limit
|
||||
/// (PHOTO_INVALID_DIMENSIONS). Verified empirically: 6300x3730 (sum 10030)
|
||||
/// fails, 6100x3900 (sum 10000) passes.
|
||||
pub const PHOTO_MAX_DIMENSION_SUM: u32 = 10000;
|
||||
/// Resize target with a safety margin so rounding cannot cross the cap.
|
||||
pub const PHOTO_TARGET_DIMENSION_SUM: u32 = 9900;
|
||||
/// Upload cap (bytes): files above this are not uploaded; the bot falls back
|
||||
/// to a smaller media URL instead.
|
||||
pub const MAX_UPLOAD_BYTES: u64 = 10 * 1024 * 1024;
|
||||
/// Decode budget (bytes): a larger intermediate buffer is not worth the peak
|
||||
/// memory; the photo degrades to the smaller URL instead.
|
||||
const MAX_DECODE_BYTES: u64 = 512 * 1024 * 1024;
|
||||
/// JPEG output quality (1-100).
|
||||
const JPEG_QUALITY: u8 = 90;
|
||||
|
||||
/// What to upload for a downloaded photo.
|
||||
pub enum PhotoPrep {
|
||||
/// Upload this file (the original when within limits, else the processed
|
||||
/// copy).
|
||||
Upload(NamedTempFile),
|
||||
/// The photo cannot be brought within Telegram's limits — the caller
|
||||
/// falls back to the item's smaller URL.
|
||||
UseFallback,
|
||||
}
|
||||
|
||||
/// A decoded image buffer tagged with its channel layout.
|
||||
#[derive(Debug)]
|
||||
enum PixBuf {
|
||||
Gray(Vec<u8>),
|
||||
GrayAlpha(Vec<u8>),
|
||||
Rgb(Vec<u8>),
|
||||
}
|
||||
|
||||
impl PixBuf {
|
||||
fn pixel_type(&self) -> fir::PixelType {
|
||||
match self {
|
||||
PixBuf::Gray(_) => fir::PixelType::U8,
|
||||
PixBuf::GrayAlpha(_) => fir::PixelType::U8x2,
|
||||
PixBuf::Rgb(_) => fir::PixelType::U8x3,
|
||||
}
|
||||
}
|
||||
|
||||
fn into_vec(self) -> Vec<u8> {
|
||||
match self {
|
||||
PixBuf::Gray(v) | PixBuf::GrayAlpha(v) | PixBuf::Rgb(v) => v,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Entry point: detects the format and processes the photo if needed.
|
||||
pub fn prepare_photo(file: NamedTempFile) -> Result<PhotoPrep, String> {
|
||||
let bytes = std::fs::read(file.path()).map_err(|e| format!("prepare read failed: {e}"))?;
|
||||
if bytes.starts_with(b"\x89PNG\r\n\x1a\n") {
|
||||
prepare_png(file, bytes)
|
||||
} else if bytes.starts_with(&[0xFF, 0xD8, 0xFF]) {
|
||||
prepare_jpeg(file, bytes)
|
||||
} else {
|
||||
log::warn!("photo in unsupported format; falling back to smaller media");
|
||||
Ok(PhotoPrep::UseFallback)
|
||||
}
|
||||
}
|
||||
|
||||
/// Parses the PNG IHDR (bytes 8..26: signature + length + "IHDR" + width +
|
||||
/// height + bit depth + color type).
|
||||
fn parse_png_header(bytes: &[u8]) -> Option<(u32, u32, png::BitDepth, png::ColorType)> {
|
||||
if !bytes.starts_with(b"\x89PNG\r\n\x1a\n") || bytes.len() < 26 {
|
||||
return None;
|
||||
}
|
||||
let w = u32::from_be_bytes(bytes.get(16..20)?.try_into().ok()?);
|
||||
let h = u32::from_be_bytes(bytes.get(20..24)?.try_into().ok()?);
|
||||
let depth = match *bytes.get(24)? {
|
||||
1 => png::BitDepth::One,
|
||||
2 => png::BitDepth::Two,
|
||||
4 => png::BitDepth::Four,
|
||||
8 => png::BitDepth::Eight,
|
||||
16 => png::BitDepth::Sixteen,
|
||||
_ => return None,
|
||||
};
|
||||
let color = match *bytes.get(25)? {
|
||||
0 => png::ColorType::Grayscale,
|
||||
2 => png::ColorType::Rgb,
|
||||
3 => png::ColorType::Indexed,
|
||||
4 => png::ColorType::GrayscaleAlpha,
|
||||
6 => png::ColorType::Rgba,
|
||||
_ => return None,
|
||||
};
|
||||
Some((w, h, depth, color))
|
||||
}
|
||||
|
||||
/// Output channels of a decoded frame for the given color type (post
|
||||
/// STRIP_16; palette expands to RGB).
|
||||
fn output_channels(color: png::ColorType) -> usize {
|
||||
match color {
|
||||
png::ColorType::Grayscale => 1,
|
||||
png::ColorType::GrayscaleAlpha => 2,
|
||||
png::ColorType::Rgb | png::ColorType::Indexed => 3,
|
||||
png::ColorType::Rgba => 4,
|
||||
}
|
||||
}
|
||||
|
||||
/// The 32→24 rule: RGBA (32-bit) becomes RGB with alpha composited onto
|
||||
/// white; 16-bit per channel was already stripped to 8-bit at decode.
|
||||
fn flatten_rgba_to_rgb(rgba: &[u8]) -> Vec<u8> {
|
||||
let mut rgb = Vec::with_capacity(rgba.len() / 4 * 3);
|
||||
for px in rgba.chunks_exact(4) {
|
||||
let a = px[3] as u32;
|
||||
for v in &px[..3] {
|
||||
// Over white: C = C*a/255 + 255*(1 - a/255).
|
||||
let v = (*v as u32 * a + 255 * (255 - a)) / 255;
|
||||
rgb.push(v.min(255) as u8);
|
||||
}
|
||||
}
|
||||
rgb
|
||||
}
|
||||
|
||||
/// Lanczos3 downsampling via fast_image_resize.
|
||||
fn resize_pix(pix: PixBuf, w: u32, h: u32, nw: u32, nh: u32) -> Result<PixBuf, String> {
|
||||
let pixel_type = pix.pixel_type();
|
||||
let src = fir::images::Image::from_vec_u8(w, h, pix.into_vec(), pixel_type)
|
||||
.map_err(|e| format!("resize input: {e}"))?;
|
||||
let mut dst = fir::images::Image::new(nw, nh, pixel_type);
|
||||
let mut resizer = fir::Resizer::new();
|
||||
let options = fir::ResizeOptions::default()
|
||||
.resize_alg(fir::ResizeAlg::Convolution(fir::FilterType::Lanczos3));
|
||||
resizer
|
||||
.resize(&src, &mut dst, &options)
|
||||
.map_err(|e| format!("resize: {e}"))?;
|
||||
let buf = dst.into_vec();
|
||||
Ok(match pixel_type {
|
||||
fir::PixelType::U8 => PixBuf::Gray(buf),
|
||||
fir::PixelType::U8x2 => PixBuf::GrayAlpha(buf),
|
||||
_ => PixBuf::Rgb(buf),
|
||||
})
|
||||
}
|
||||
|
||||
fn encode_png(out: &mut Vec<u8>, pix: &PixBuf, w: u32, h: u32) -> Result<(), png::EncodingError> {
|
||||
let (color, buf) = match pix {
|
||||
PixBuf::Gray(v) => (png::ColorType::Grayscale, v.as_slice()),
|
||||
PixBuf::GrayAlpha(v) => (png::ColorType::GrayscaleAlpha, v.as_slice()),
|
||||
PixBuf::Rgb(v) => (png::ColorType::Rgb, v.as_slice()),
|
||||
};
|
||||
let mut encoder = png::Encoder::new(out, w, h);
|
||||
encoder.set_color(color);
|
||||
encoder.set_depth(png::BitDepth::Eight);
|
||||
let mut writer = encoder.write_header()?;
|
||||
writer.write_image_data(buf)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn encode_jpeg(pix: &PixBuf, w: u32, h: u32) -> Result<Vec<u8>, String> {
|
||||
use jpeg_encoder::{ColorType, Encoder};
|
||||
let mut out = Vec::new();
|
||||
let encoder = Encoder::new(&mut out, JPEG_QUALITY);
|
||||
match pix {
|
||||
PixBuf::Gray(v) => encoder
|
||||
.encode(v, w as u16, h as u16, ColorType::Luma)
|
||||
.map_err(|e| format!("jpeg encode: {e}"))?,
|
||||
PixBuf::GrayAlpha(v) => {
|
||||
// JPEG has no alpha: composite onto white, output as gray.
|
||||
let gray: Vec<u8> = v
|
||||
.chunks_exact(2)
|
||||
.map(|px| {
|
||||
let (g, a) = (px[0] as u32, px[1] as u32);
|
||||
((g * a + 255 * (255 - a)) / 255).min(255) as u8
|
||||
})
|
||||
.collect();
|
||||
encoder
|
||||
.encode(&gray, w as u16, h as u16, ColorType::Luma)
|
||||
.map_err(|e| format!("jpeg encode: {e}"))?;
|
||||
}
|
||||
PixBuf::Rgb(v) => encoder
|
||||
.encode(v, w as u16, h as u16, ColorType::Rgb)
|
||||
.map_err(|e| format!("jpeg encode: {e}"))?,
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
fn write_temp(bytes: &[u8], ext: &str) -> Result<NamedTempFile, String> {
|
||||
let mut file = tempfile::Builder::new()
|
||||
.suffix(&format!(".{ext}"))
|
||||
.tempfile()
|
||||
.map_err(|e| format!("temp file failed: {e}"))?;
|
||||
file.as_file_mut()
|
||||
.write_all(bytes)
|
||||
.map_err(|e| format!("temp file write failed: {e}"))?;
|
||||
Ok(file)
|
||||
}
|
||||
|
||||
fn target_dims(w: u32, h: u32) -> (u32, u32) {
|
||||
let scale = PHOTO_TARGET_DIMENSION_SUM as f64 / (w + h) as f64;
|
||||
(
|
||||
((w as f64 * scale).round() as u32).max(1),
|
||||
((h as f64 * scale).round() as u32).max(1),
|
||||
)
|
||||
}
|
||||
|
||||
/// PNG branch: decode (16→8, palette→RGB; gray/GA stay), flatten RGBA to
|
||||
/// RGB, Lanczos-downscale beyond the dimension cap, encode PNG — a PNG still
|
||||
/// over the upload cap afterwards becomes JPEG.
|
||||
fn prepare_png(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String> {
|
||||
let (w, h, _bit_depth, color_type) =
|
||||
parse_png_header(&bytes).ok_or("invalid PNG header")?;
|
||||
let size_over = bytes.len() as u64 > MAX_UPLOAD_BYTES;
|
||||
if w + h <= PHOTO_MAX_DIMENSION_SUM && !size_over {
|
||||
return Ok(PhotoPrep::Upload(file));
|
||||
}
|
||||
log::info!("photo {w}x{h} ({_bit_depth:?} {color_type:?}, {} bytes) needs processing", bytes.len());
|
||||
|
||||
let channels = output_channels(color_type);
|
||||
if (w as u64) * (h as u64) * channels as u64 > MAX_DECODE_BYTES {
|
||||
log::warn!("photo decode buffer exceeds the memory budget; falling back to smaller media");
|
||||
return Ok(PhotoPrep::UseFallback);
|
||||
}
|
||||
|
||||
// STRIP_16 drops 16-bit to 8-bit (the depth-reduction step); palette
|
||||
// expands to RGB (resampling requires it). Gray and gray-alpha are kept.
|
||||
let transforms = match color_type {
|
||||
png::ColorType::Indexed => png::Transformations::EXPAND,
|
||||
_ => png::Transformations::STRIP_16,
|
||||
};
|
||||
let mut decoder = png::Decoder::new(std::io::Cursor::new(&bytes));
|
||||
decoder.set_transformations(transforms);
|
||||
let mut reader = decoder.read_info().map_err(|e| format!("png decode: {e}"))?;
|
||||
let out_w = reader.info().width;
|
||||
let out_h = reader.info().height;
|
||||
let mut buf = vec![
|
||||
0u8;
|
||||
reader
|
||||
.output_buffer_size()
|
||||
.ok_or("png output buffer size")?
|
||||
];
|
||||
reader
|
||||
.next_frame(&mut buf)
|
||||
.map_err(|e| format!("png frame: {e}"))?;
|
||||
|
||||
let mut pix = match color_type {
|
||||
png::ColorType::Rgba => PixBuf::Rgb(flatten_rgba_to_rgb(&buf)),
|
||||
png::ColorType::Grayscale => PixBuf::Gray(buf),
|
||||
png::ColorType::GrayscaleAlpha => PixBuf::GrayAlpha(buf),
|
||||
png::ColorType::Rgb | png::ColorType::Indexed => PixBuf::Rgb(buf),
|
||||
};
|
||||
|
||||
let (mut w, mut h) = (out_w, out_h);
|
||||
if w + h > PHOTO_MAX_DIMENSION_SUM {
|
||||
let (nw, nh) = target_dims(w, h);
|
||||
pix = resize_pix(pix, w, h, nw, nh)?;
|
||||
(w, h) = (nw, nh);
|
||||
log::info!("downscaled photo to {w}x{h} (Lanczos3)");
|
||||
}
|
||||
|
||||
let mut png_bytes = Vec::new();
|
||||
encode_png(&mut png_bytes, &pix, w, h).map_err(|e| format!("png encode: {e}"))?;
|
||||
if png_bytes.len() as u64 <= MAX_UPLOAD_BYTES {
|
||||
return Ok(PhotoPrep::Upload(write_temp(&png_bytes, "png")?));
|
||||
}
|
||||
log::info!("PNG still over the upload cap after processing; transcoding to JPEG");
|
||||
let jpeg_bytes = encode_jpeg(&pix, w, h)?;
|
||||
if jpeg_bytes.len() as u64 <= MAX_UPLOAD_BYTES {
|
||||
return Ok(PhotoPrep::Upload(write_temp(&jpeg_bytes, "jpg")?));
|
||||
}
|
||||
log::warn!("processed photo still exceeds the upload cap; falling back to smaller media");
|
||||
Ok(PhotoPrep::UseFallback)
|
||||
}
|
||||
|
||||
/// JPEG branch: zune-jpeg decode → Lanczos downscale → jpeg-encoder output.
|
||||
fn prepare_jpeg(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String> {
|
||||
let mut decoder = zune_jpeg::JpegDecoder::new(std::io::Cursor::new(&bytes));
|
||||
// Decodes to RGB by default. Headers first so dimensions are known before
|
||||
// the (potentially huge) pixel decode.
|
||||
decoder
|
||||
.decode_headers()
|
||||
.map_err(|e| format!("jpeg headers: {e}"))?;
|
||||
let info = decoder.info().ok_or("jpeg info unavailable")?;
|
||||
let (w, h) = (info.width as u32, info.height as u32);
|
||||
let size_over = bytes.len() as u64 > MAX_UPLOAD_BYTES;
|
||||
if w + h <= PHOTO_MAX_DIMENSION_SUM && !size_over {
|
||||
return Ok(PhotoPrep::Upload(file));
|
||||
}
|
||||
if (w as u64) * (h as u64) * 3 > MAX_DECODE_BYTES {
|
||||
log::warn!("photo decode buffer exceeds the memory budget; falling back to smaller media");
|
||||
return Ok(PhotoPrep::UseFallback);
|
||||
}
|
||||
let pixels = decoder.decode().map_err(|e| format!("jpeg decode: {e}"))?;
|
||||
let mut pix = PixBuf::Rgb(pixels);
|
||||
let (mut w, mut h) = (w, h);
|
||||
if w + h > PHOTO_MAX_DIMENSION_SUM {
|
||||
let (nw, nh) = target_dims(w, h);
|
||||
pix = resize_pix(pix, w, h, nw, nh)?;
|
||||
(w, h) = (nw, nh);
|
||||
log::info!("downscaled jpeg to {w}x{h} (Lanczos3)");
|
||||
}
|
||||
let jpeg_bytes = encode_jpeg(&pix, w, h)?;
|
||||
if jpeg_bytes.len() as u64 <= MAX_UPLOAD_BYTES {
|
||||
return Ok(PhotoPrep::Upload(write_temp(&jpeg_bytes, "jpg")?));
|
||||
}
|
||||
log::warn!("processed photo still exceeds the upload cap; falling back to smaller media");
|
||||
Ok(PhotoPrep::UseFallback)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn png_header(w: u32, h: u32, depth: u8, color: u8) -> Vec<u8> {
|
||||
let mut bytes = b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR".to_vec();
|
||||
bytes.extend(w.to_be_bytes());
|
||||
bytes.extend(h.to_be_bytes());
|
||||
bytes.extend([depth, color, 0, 0, 0]);
|
||||
bytes
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_png_header() {
|
||||
let bytes = png_header(8979, 5316, 16, 6); // 16-bit RGBA
|
||||
let (w, h, depth, color) = parse_png_header(&bytes).unwrap();
|
||||
assert_eq!((w, h), (8979, 5316));
|
||||
assert_eq!(depth, png::BitDepth::Sixteen);
|
||||
assert_eq!(color, png::ColorType::Rgba);
|
||||
|
||||
let (_, _, depth, color) = parse_png_header(&png_header(10, 10, 8, 0)).unwrap();
|
||||
assert_eq!(depth, png::BitDepth::Eight);
|
||||
assert_eq!(color, png::ColorType::Grayscale);
|
||||
|
||||
assert!(parse_png_header(b"not a png").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn flatten_rgba_to_rgb_composites_over_white() {
|
||||
// opaque red stays red
|
||||
assert_eq!(flatten_rgba_to_rgb(&[255, 0, 0, 255]), vec![255, 0, 0]);
|
||||
// fully transparent → white
|
||||
assert_eq!(flatten_rgba_to_rgb(&[0, 0, 0, 0]), vec![255, 255, 255]);
|
||||
// half alpha red → (255+255)/2 = 255, (0*128 + 255*127)/255 = 127
|
||||
let out = flatten_rgba_to_rgb(&[255, 0, 0, 128]);
|
||||
assert_eq!(out[0], 255);
|
||||
assert_eq!(out[1], 127);
|
||||
assert_eq!(out[2], 127);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn target_dims_stay_under_the_cap() {
|
||||
for (w, h) in [(12000u32, 7000u32), (10000, 10000), (8979, 5316)] {
|
||||
let (nw, nh) = target_dims(w, h);
|
||||
assert!(nw + nh <= PHOTO_MAX_DIMENSION_SUM, "{w}x{h} -> {nw}x{nh}");
|
||||
assert!(nw >= 1 && nh >= 1);
|
||||
}
|
||||
// already within limits: no change expected from the caller, but the
|
||||
// helper must not produce zero dimensions.
|
||||
let (nw, nh) = target_dims(500, 400);
|
||||
assert!(nw >= 1 && nh >= 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_pix_changes_dimensions() {
|
||||
// 300x200 RGB → 100x66
|
||||
let buf: Vec<u8> = (0..300 * 200 * 3).map(|i| (i % 251) as u8).collect();
|
||||
let resized = resize_pix(PixBuf::Rgb(buf), 300, 200, 100, 66).unwrap();
|
||||
match resized {
|
||||
PixBuf::Rgb(v) => assert_eq!(v.len(), 100 * 66 * 3),
|
||||
other => panic!("expected rgb, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn png_encode_roundtrip_keeps_gray() {
|
||||
let gray = vec![128u8; 4 * 4];
|
||||
let mut out = Vec::new();
|
||||
encode_png(&mut out, &PixBuf::Gray(gray), 4, 4).unwrap();
|
||||
assert!(!out.is_empty());
|
||||
let (_, _, depth, color) = parse_png_header(&out).unwrap();
|
||||
assert_eq!(depth, png::BitDepth::Eight);
|
||||
assert_eq!(color, png::ColorType::Grayscale);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn jpeg_encode_produces_bytes() {
|
||||
let rgb = vec![128u8; 8 * 8 * 3];
|
||||
let out = encode_jpeg(&PixBuf::Rgb(rgb), 8, 8).unwrap();
|
||||
assert!(out.len() > 100);
|
||||
assert!(out.starts_with(&[0xFF, 0xD8]));
|
||||
}
|
||||
|
||||
/// Writes a small dimension-oversized PNG (9999x2 → sum 10001) to a temp
|
||||
/// file and runs the full pipeline.
|
||||
fn run_pipeline(w: u32, h: u32, color: png::ColorType, fill: u8) -> Result<PhotoPrep, String> {
|
||||
let (channels, data): (usize, Vec<u8>) = match color {
|
||||
png::ColorType::Grayscale => (1, vec![fill; (w * h) as usize]),
|
||||
png::ColorType::Rgb => (3, vec![fill; (w * h * 3) as usize]),
|
||||
_ => unreachable!(),
|
||||
};
|
||||
let mut bytes = Vec::new();
|
||||
{
|
||||
let mut encoder = png::Encoder::new(&mut bytes, w, h);
|
||||
encoder.set_color(color);
|
||||
encoder.set_depth(png::BitDepth::Eight);
|
||||
let mut writer = encoder.write_header().unwrap();
|
||||
writer.write_image_data(&data).unwrap();
|
||||
}
|
||||
assert_eq!(data.len(), channels * (w * h) as usize);
|
||||
|
||||
let mut file = tempfile::Builder::new().suffix(".png").tempfile().unwrap();
|
||||
std::io::Write::write_all(file.as_file_mut(), &bytes).unwrap();
|
||||
prepare_photo(file)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pipeline_downscales_oversized_png_keeping_format() {
|
||||
let prep = run_pipeline(9999, 2, png::ColorType::Rgb, 128).unwrap();
|
||||
match prep {
|
||||
PhotoPrep::Upload(file) => {
|
||||
let out = std::fs::read(file.path()).unwrap();
|
||||
let (w, h, depth, color) = parse_png_header(&out).unwrap();
|
||||
assert!(w + h <= PHOTO_MAX_DIMENSION_SUM, "{w}x{h}");
|
||||
assert_eq!(depth, png::BitDepth::Eight);
|
||||
assert_eq!(color, png::ColorType::Rgb);
|
||||
}
|
||||
PhotoPrep::UseFallback => panic!("over-dimension PNG should have been resized"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pipeline_keeps_gray_png_gray() {
|
||||
let prep = run_pipeline(9999, 2, png::ColorType::Grayscale, 200).unwrap();
|
||||
match prep {
|
||||
PhotoPrep::Upload(file) => {
|
||||
let out = std::fs::read(file.path()).unwrap();
|
||||
let (_, _, _, color) = parse_png_header(&out).unwrap();
|
||||
assert_eq!(color, png::ColorType::Grayscale, "gray must not upconvert");
|
||||
}
|
||||
PhotoPrep::UseFallback => panic!("over-dimension gray PNG should have been resized"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pipeline_resizes_oversized_jpeg() {
|
||||
// Build a small over-dimension JPEG with jpeg-encoder.
|
||||
let (w, h) = (9999u16, 2u16);
|
||||
let rgb = vec![90u8; (w as usize) * (h as usize) * 3];
|
||||
let mut bytes = Vec::new();
|
||||
{
|
||||
let encoder = jpeg_encoder::Encoder::new(&mut bytes, 90);
|
||||
encoder.encode(&rgb, w, h, jpeg_encoder::ColorType::Rgb).unwrap();
|
||||
}
|
||||
let mut file = tempfile::Builder::new().suffix(".jpg").tempfile().unwrap();
|
||||
std::io::Write::write_all(file.as_file_mut(), &bytes).unwrap();
|
||||
match prepare_photo(file).unwrap() {
|
||||
PhotoPrep::Upload(file) => {
|
||||
let out = std::fs::read(file.path()).unwrap();
|
||||
assert!(out.starts_with(&[0xFF, 0xD8]), "output must stay jpeg");
|
||||
// 9999x2 downscaled: the buffer length tells the new dims.
|
||||
assert!(out.len() > 100);
|
||||
}
|
||||
PhotoPrep::UseFallback => panic!("over-dimension JPEG should have been resized"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "heavy: generates a >10 MiB PNG (run explicitly)"]
|
||||
fn pipeline_transcodes_oversized_png_to_jpeg() {
|
||||
// 6000x4000 (sum 10000 — under the dimension cap) smooth gradient with
|
||||
// small per-pixel noise: PNG-incompressible (delta filters defeated)
|
||||
// but JPEG-friendly (DCT smooths the small noise). Verified with
|
||||
// ffmpeg: 8000x6000 amp-5 variant is a 59 MB PNG / 3.3 MB JPEG.
|
||||
let (w, h) = (6000u32, 4000u32);
|
||||
let mut rng = 0x1234_5678_9abc_def0u64;
|
||||
let mut data = Vec::with_capacity((w * h * 3) as usize);
|
||||
for y in 0..h {
|
||||
for x in 0..w {
|
||||
let base = (x + y) * 255 / (w + h);
|
||||
rng = rng.wrapping_mul(6364136223846793005).wrapping_add(1442695040888963407);
|
||||
let n = ((rng >> 33) % 11) as i32 - 5; // noise in [-5, 5]
|
||||
let v = (base as i32 + n).clamp(0, 255) as u8;
|
||||
data.extend_from_slice(&[v, v, v]);
|
||||
}
|
||||
}
|
||||
let mut bytes = Vec::new();
|
||||
{
|
||||
let mut encoder = png::Encoder::new(&mut bytes, w, h);
|
||||
encoder.set_color(png::ColorType::Rgb);
|
||||
encoder.set_depth(png::BitDepth::Eight);
|
||||
let mut writer = encoder.write_header().unwrap();
|
||||
writer.write_image_data(&data).unwrap();
|
||||
}
|
||||
assert!(bytes.len() as u64 > MAX_UPLOAD_BYTES, "test needs a >10MiB PNG, got {}", bytes.len());
|
||||
|
||||
let mut file = tempfile::Builder::new().suffix(".png").tempfile().unwrap();
|
||||
std::io::Write::write_all(file.as_file_mut(), &bytes).unwrap();
|
||||
match prepare_photo(file).unwrap() {
|
||||
PhotoPrep::Upload(file) => {
|
||||
let out = std::fs::read(file.path()).unwrap();
|
||||
assert!(out.starts_with(&[0xFF, 0xD8]), "must transcode to JPEG");
|
||||
assert!(out.len() as u64 <= MAX_UPLOAD_BYTES);
|
||||
}
|
||||
PhotoPrep::UseFallback => panic!("PNG over the byte cap must transcode to JPEG"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,7 +6,7 @@
|
||||
//! replaced by dedicated columns.
|
||||
|
||||
use parking_lot::Mutex;
|
||||
use rusqlite::{params, Connection};
|
||||
use rusqlite::{params, Connection, TransactionBehavior};
|
||||
use serde_json::Value;
|
||||
use std::pin::Pin;
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
@@ -18,6 +18,12 @@ use tokio::task::JoinHandle;
|
||||
pub const MAX_RETRIES: u32 = 2;
|
||||
pub const LOCK_TTL_SECONDS: f64 = 120.0;
|
||||
|
||||
/// Number of concurrent worker loops. Tasks are independent (retries and
|
||||
/// forward resumes); leases serialize row claims via SQLite transactions, so
|
||||
/// extra workers drain backlogs faster. Each worker can be mid-send to
|
||||
/// Telegram at the same time as handler tasks, so keep this modest.
|
||||
const QUEUE_WORKERS: usize = 4;
|
||||
|
||||
/// What a handler returns instead of throwing. The payload it carries is the
|
||||
/// (possibly updated) task state to persist for the next attempt.
|
||||
pub enum QueueError {
|
||||
@@ -42,7 +48,7 @@ pub struct PersistentTaskQueue {
|
||||
db_path: String,
|
||||
notify: Arc<Notify>,
|
||||
stop: Arc<AtomicBool>,
|
||||
worker: Mutex<Option<JoinHandle<()>>>,
|
||||
worker: Mutex<Vec<JoinHandle<()>>>,
|
||||
counter: AtomicU64,
|
||||
}
|
||||
|
||||
@@ -68,6 +74,15 @@ fn now_f64() -> f64 {
|
||||
.unwrap_or(0.0)
|
||||
}
|
||||
|
||||
/// Opens the queue DB with a busy timeout. Handler tasks enqueue while
|
||||
/// workers lease/update rows concurrently; without the timeout a concurrent
|
||||
/// write fails immediately with SQLITE_BUSY and the operation is lost.
|
||||
fn open_db(path: &str) -> rusqlite::Result<Connection> {
|
||||
let conn = Connection::open(path)?;
|
||||
conn.busy_timeout(Duration::from_secs(5))?;
|
||||
Ok(conn)
|
||||
}
|
||||
|
||||
fn ensure_schema(conn: &Connection) -> rusqlite::Result<()> {
|
||||
conn.execute_batch(
|
||||
"CREATE TABLE IF NOT EXISTS tasks (id TEXT PRIMARY KEY, payload TEXT NOT NULL, \
|
||||
@@ -96,12 +111,12 @@ impl PersistentTaskQueue {
|
||||
db_path: db_path.to_string(),
|
||||
notify: Arc::new(Notify::new()),
|
||||
stop: Arc::new(AtomicBool::new(false)),
|
||||
worker: Mutex::new(None),
|
||||
worker: Mutex::new(Vec::new()),
|
||||
counter: AtomicU64::new(0),
|
||||
}
|
||||
}
|
||||
|
||||
/// Starts the worker loop. Also recovers rows left `in_progress` by a
|
||||
/// Starts the worker loops. Also recovers rows left `in_progress` by a
|
||||
/// previous process (lease expired).
|
||||
pub async fn start<H, F, D, G>(&self, handler: H, dead_letter: D)
|
||||
where
|
||||
@@ -114,21 +129,25 @@ impl PersistentTaskQueue {
|
||||
let dead_letter: Arc<DeadLetter> =
|
||||
Arc::new(move |payload, message| Box::pin(dead_letter(payload, message)));
|
||||
self.recover_stale().await;
|
||||
let worker = QueueWorker {
|
||||
db_path: self.db_path.clone(),
|
||||
notify: Arc::clone(&self.notify),
|
||||
stop: Arc::clone(&self.stop),
|
||||
handler,
|
||||
dead_letter,
|
||||
};
|
||||
let worker = tokio::spawn(worker.run_loop());
|
||||
*self.worker.lock() = Some(worker);
|
||||
let mut handles = Vec::with_capacity(QUEUE_WORKERS);
|
||||
for _ in 0..QUEUE_WORKERS {
|
||||
let worker = QueueWorker {
|
||||
db_path: self.db_path.clone(),
|
||||
notify: Arc::clone(&self.notify),
|
||||
stop: Arc::clone(&self.stop),
|
||||
handler: Arc::clone(&handler),
|
||||
dead_letter: Arc::clone(&dead_letter),
|
||||
};
|
||||
handles.push(tokio::spawn(worker.run_loop()));
|
||||
}
|
||||
*self.worker.lock() = handles;
|
||||
}
|
||||
|
||||
pub async fn stop(&self) {
|
||||
self.stop.store(true, Ordering::Relaxed);
|
||||
self.notify.notify_one();
|
||||
if let Some(handle) = self.worker.lock().take() {
|
||||
self.notify.notify_waiters();
|
||||
let handles = std::mem::take(&mut *self.worker.lock());
|
||||
for handle in handles {
|
||||
let _ = handle.await;
|
||||
}
|
||||
}
|
||||
@@ -145,8 +164,8 @@ impl PersistentTaskQueue {
|
||||
let payload = payload.to_string();
|
||||
let db_path = self.db_path.clone();
|
||||
log::info!("enqueued {id} (run_after {run_after:.1})");
|
||||
let result = tokio::task::spawn_blocking(move || -> rusqlite::Result<()> {
|
||||
let conn = Connection::open(&db_path)?;
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<()> {
|
||||
let conn = open_db(&db_path)?;
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO tasks (id, payload, run_after, attempts, status, locked_until, created_at) \
|
||||
VALUES (?1, ?2, ?3, 0, 'pending', 0, ?4)",
|
||||
@@ -156,14 +175,16 @@ impl PersistentTaskQueue {
|
||||
})
|
||||
.await
|
||||
.expect("queue insert worker panicked")?;
|
||||
self.notify.notify_one();
|
||||
Ok(result)
|
||||
// Wake every sleeping worker: with several workers the one that finds
|
||||
// nothing due must not starve the newly inserted row.
|
||||
self.notify.notify_waiters();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn recover_stale(&self) {
|
||||
let db_path = self.db_path.clone();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<()> {
|
||||
let conn = Connection::open(&db_path)?;
|
||||
let conn = open_db(&db_path)?;
|
||||
conn.execute(
|
||||
"UPDATE tasks SET status='pending', locked_until=0 WHERE status='in_progress' AND locked_until < ?1",
|
||||
params![now_f64()],
|
||||
@@ -206,11 +227,15 @@ impl QueueWorker {
|
||||
async fn lease_next(&self) -> Option<LeasedRow> {
|
||||
let db_path = self.db_path.clone();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<Option<LeasedRow>> {
|
||||
let mut conn = Connection::open(&db_path)?;
|
||||
let tx = conn.transaction()?;
|
||||
let mut conn = open_db(&db_path)?;
|
||||
// BEGIN IMMEDIATE: with several workers, a deferred transaction
|
||||
// that read before another worker's lease commit would fail with
|
||||
// SQLITE_BUSY_SNAPSHOT. Taking the write lock up front serializes
|
||||
// leases and re-reads the freshest committed state.
|
||||
let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?;
|
||||
let now = now_f64();
|
||||
let row = tx.query_row(
|
||||
"SELECT id, payload, attempts FROM tasks WHERE status='pending' AND run_after <= ?1 \
|
||||
"SELECT id, payload, attempts FROM tasks WHERE status='pending' AND run_after <= ?1 AND locked_until <= ?1 \
|
||||
ORDER BY run_after LIMIT 1",
|
||||
params![now],
|
||||
|r| {
|
||||
@@ -251,7 +276,7 @@ impl QueueWorker {
|
||||
async fn earliest_run_after(&self) -> Option<f64> {
|
||||
let db_path = self.db_path.clone();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<Option<f64>> {
|
||||
let conn = Connection::open(&db_path)?;
|
||||
let conn = open_db(&db_path)?;
|
||||
let mut stmt = conn.prepare("SELECT MIN(run_after) FROM tasks WHERE status='pending'")?;
|
||||
let mut rows = stmt.query([])?;
|
||||
match rows.next()? {
|
||||
@@ -314,7 +339,7 @@ impl QueueWorker {
|
||||
let db_path = self.db_path.clone();
|
||||
let id = id.to_string();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<()> {
|
||||
let conn = Connection::open(&db_path)?;
|
||||
let conn = open_db(&db_path)?;
|
||||
conn.execute("DELETE FROM tasks WHERE id = ?1", params![id])?;
|
||||
Ok(())
|
||||
})
|
||||
@@ -328,7 +353,7 @@ impl QueueWorker {
|
||||
let id = id.to_string();
|
||||
let payload = payload.to_string();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<()> {
|
||||
let conn = Connection::open(&db_path)?;
|
||||
let conn = open_db(&db_path)?;
|
||||
conn.execute(
|
||||
"UPDATE tasks SET payload=?1, run_after=?2, attempts=?3, status='pending', locked_until=0 WHERE id=?4",
|
||||
params![payload, now_f64() + delay_seconds, attempts, id],
|
||||
@@ -338,7 +363,7 @@ impl QueueWorker {
|
||||
.await
|
||||
.expect("queue reschedule worker panicked")
|
||||
.unwrap_or_else(|e| log::error!("queue reschedule failed: {e}"));
|
||||
self.notify.notify_one();
|
||||
self.notify.notify_waiters();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+453
-54
@@ -3,7 +3,9 @@
|
||||
//! URL is blocked by hotlink protection; the bot downloads the file itself
|
||||
//! and uploads it via multipart).
|
||||
|
||||
use crate::handlers::{CHAT_STORE, TASK_QUEUE};
|
||||
use crate::handlers::{CHAT_STORE, LINK_CACHE, TASK_QUEUE};
|
||||
use crate::link_cache::{CachedMedia, CachedMediaKind, CachedPost};
|
||||
use crate::photo::{self, PhotoPrep, MAX_UPLOAD_BYTES};
|
||||
use crate::queue::QueueError;
|
||||
use crate::state::{EditMessage, unix_now};
|
||||
use rand::Rng;
|
||||
@@ -25,18 +27,43 @@ pub enum MediaItemPayload {
|
||||
Photo {
|
||||
media: String,
|
||||
has_spoiler: bool,
|
||||
/// Smaller variant used when the primary media exceeds Telegram's
|
||||
/// size limits.
|
||||
#[serde(default)]
|
||||
fallback_url: Option<String>,
|
||||
/// `media` is a Telegram file id (link-cache hit), not a URL.
|
||||
#[serde(default)]
|
||||
file_id: bool,
|
||||
},
|
||||
Video {
|
||||
media: String,
|
||||
has_spoiler: bool,
|
||||
thumbnail: Option<String>,
|
||||
#[serde(default)]
|
||||
fallback_url: Option<String>,
|
||||
/// `media` is a Telegram file id (link-cache hit), not a URL.
|
||||
#[serde(default)]
|
||||
file_id: bool,
|
||||
},
|
||||
Animation {
|
||||
media: String,
|
||||
has_spoiler: bool,
|
||||
/// `media` is a Telegram file id (link-cache hit), not a URL.
|
||||
#[serde(default)]
|
||||
file_id: bool,
|
||||
},
|
||||
}
|
||||
|
||||
impl MediaItemPayload {
|
||||
fn fallback_url(&self) -> Option<&str> {
|
||||
match self {
|
||||
MediaItemPayload::Photo { fallback_url, .. }
|
||||
| MediaItemPayload::Video { fallback_url, .. } => fallback_url.as_deref(),
|
||||
MediaItemPayload::Animation { .. } => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize, Clone, Debug)]
|
||||
#[serde(tag = "type", rename_all = "snake_case")]
|
||||
pub enum Task {
|
||||
@@ -52,6 +79,10 @@ pub enum Task {
|
||||
forward_channel_id: Option<i64>,
|
||||
notify_chat_id: Option<i64>,
|
||||
notify_message_id: Option<i64>,
|
||||
/// Raw render data captured on a cache miss; the send fills in the
|
||||
/// Telegram file ids and persists the entry (see `link_cache`).
|
||||
#[serde(default)]
|
||||
cache_data: Option<CachedPost>,
|
||||
},
|
||||
SendAnimation {
|
||||
chat_id: i64,
|
||||
@@ -63,6 +94,10 @@ pub enum Task {
|
||||
forward_channel_id: Option<i64>,
|
||||
notify_chat_id: Option<i64>,
|
||||
notify_message_id: Option<i64>,
|
||||
/// Raw render data captured on a cache miss; the send fills in the
|
||||
/// Telegram file id and persists the entry (see `link_cache`).
|
||||
#[serde(default)]
|
||||
cache_data: Option<CachedPost>,
|
||||
},
|
||||
ForwardMessages {
|
||||
from_chat_id: i64,
|
||||
@@ -73,8 +108,109 @@ pub enum Task {
|
||||
},
|
||||
}
|
||||
|
||||
impl Task {
|
||||
fn cache_data(&self) -> Option<&CachedPost> {
|
||||
match self {
|
||||
Task::SendMediaSequence { cache_data, .. }
|
||||
| Task::SendAnimation { cache_data, .. } => cache_data.as_ref(),
|
||||
Task::ForwardMessages { .. } => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn source_url(&self) -> Option<&str> {
|
||||
match self {
|
||||
Task::SendMediaSequence { source_url, .. }
|
||||
| Task::SendAnimation { source_url, .. } => Some(source_url),
|
||||
Task::ForwardMessages { .. } => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// True when the media payloads are Telegram file ids from the link cache
|
||||
/// (a cached file id that goes permanently bad should be dropped so the
|
||||
/// next request re-fetches).
|
||||
fn is_cached_send(&self) -> bool {
|
||||
self.cache_data().is_some_and(|c| !c.media.is_empty())
|
||||
}
|
||||
}
|
||||
|
||||
/// Telegram file id of the message's media, matched to the payload kind.
|
||||
fn file_id_of_message(message: &Message, item: &MediaItemPayload) -> Option<String> {
|
||||
match item {
|
||||
// `photo()` returns all sizes, smallest first — the largest carries
|
||||
// the file id of the sent media.
|
||||
MediaItemPayload::Photo { .. } => {
|
||||
message.photo().and_then(|sizes| sizes.last()).map(|p| p.file.id.to_string())
|
||||
}
|
||||
MediaItemPayload::Video { .. } => message.video().map(|v| v.file.id.to_string()),
|
||||
MediaItemPayload::Animation { .. } => message.animation().map(|a| a.file.id.to_string()),
|
||||
}
|
||||
}
|
||||
|
||||
fn kind_of_item(item: &MediaItemPayload) -> CachedMediaKind {
|
||||
match item {
|
||||
MediaItemPayload::Photo { .. } => CachedMediaKind::Photo,
|
||||
MediaItemPayload::Video { .. } => CachedMediaKind::Video,
|
||||
MediaItemPayload::Animation { .. } => CachedMediaKind::Animation,
|
||||
}
|
||||
}
|
||||
|
||||
/// Collects the Telegram file ids of a sent media group, aligned to the
|
||||
/// batch's items.
|
||||
fn collect_file_ids(messages: &[Message], batch: &[MediaItemPayload], out: &mut Vec<CachedMedia>) {
|
||||
for (message, item) in messages.iter().zip(batch.iter()) {
|
||||
if let Some(file_id) = file_id_of_message(message, item) {
|
||||
out.push(CachedMedia {
|
||||
kind: kind_of_item(item),
|
||||
file_id,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Persists a successful send under the post's cache key. Only runs for a
|
||||
/// fresh (non-resumed) task that carried raw cache data with no file ids yet.
|
||||
async fn cache_sent_task(task: &Task, media: Vec<CachedMedia>) {
|
||||
let Some(cache_data) = task.cache_data() else {
|
||||
return;
|
||||
};
|
||||
if !cache_data.media.is_empty() || media.is_empty() {
|
||||
return;
|
||||
}
|
||||
let mut post = cache_data.clone();
|
||||
post.media = media;
|
||||
if let Some(key) = x_media::site::cache_key(&post.url) {
|
||||
LINK_CACHE.put(&key, &post).await;
|
||||
log::info!("cached send for {}", post.url);
|
||||
}
|
||||
}
|
||||
|
||||
/// Persists a lone animation send under the post's cache key.
|
||||
async fn cache_animation_send(task: &Task, message: &Message) {
|
||||
if let Some(file_id) = message.animation().map(|a| a.file.id.to_string()) {
|
||||
cache_sent_task(
|
||||
task,
|
||||
vec![CachedMedia {
|
||||
kind: CachedMediaKind::Animation,
|
||||
file_id,
|
||||
}],
|
||||
)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
|
||||
/// A cached Telegram file id failed permanently (stale/expired); drop the
|
||||
/// cache entry so the next request re-fetches instead of repeating it.
|
||||
pub async fn invalidate_cache(task: &Task) {
|
||||
if task.is_cached_send()
|
||||
&& let Some(url) = task.source_url()
|
||||
&& let Some(key) = x_media::site::cache_key(url)
|
||||
{
|
||||
log::info!("removing stale link cache entry for {url}");
|
||||
LINK_CACHE.remove(&key).await;
|
||||
}
|
||||
}
|
||||
|
||||
pub const MAX_MEDIA_GROUP: usize = 9;
|
||||
pub const MAX_UPLOAD_BYTES: u64 = 50 * 1024 * 1024; // Telegram Bot API upload cap
|
||||
|
||||
/// Splits media into batches of at most [`MAX_MEDIA_GROUP`] items.
|
||||
pub fn chunk_media_items<T: Clone>(items: Vec<T>) -> Vec<Vec<T>> {
|
||||
@@ -91,17 +227,32 @@ pub fn retry_delay_seconds(attempts: u32) -> f64 {
|
||||
/// these errors are handled by the download-and-reupload fallback, NOT by a
|
||||
/// queue retry (resending the URL cannot succeed).
|
||||
pub fn is_media_fetch_failure(e: &ApiError) -> bool {
|
||||
const MARKERS: [&str; 5] = [
|
||||
const MARKERS: [&str; 6] = [
|
||||
"webpage_media_empty",
|
||||
"media_empty",
|
||||
"empty_web_media",
|
||||
"webpage_curl_failed",
|
||||
"timeout",
|
||||
// Oversized photos (width + height > 10000 px) are rejected on URL
|
||||
// sends too; route them to the download-and-resize fallback.
|
||||
"PHOTO_INVALID_DIMENSIONS",
|
||||
];
|
||||
let description = e.to_string().to_lowercase();
|
||||
MARKERS.iter().any(|marker| description.contains(marker))
|
||||
}
|
||||
|
||||
/// Telegram reported the media file as too large (HTTP 413 on multipart
|
||||
/// upload, or a "too large" message for URL-fetched media). These errors are
|
||||
/// handled by the size-check fallback (use a smaller media URL), NOT by a
|
||||
/// queue retry.
|
||||
pub fn is_size_error(e: &ApiError) -> bool {
|
||||
if matches!(e, ApiError::RequestEntityTooLarge) {
|
||||
return true;
|
||||
}
|
||||
let description = e.to_string().to_lowercase();
|
||||
["too large", "too big"].iter().any(|marker| description.contains(marker))
|
||||
}
|
||||
|
||||
/// Task-free classification of a Telegram request error. The callers attach
|
||||
/// the (updated) task when building a [`SendError`].
|
||||
pub enum Classification {
|
||||
@@ -168,6 +319,32 @@ fn input_file_for(media: &str) -> Result<InputFile, String> {
|
||||
}
|
||||
}
|
||||
|
||||
impl MediaItemPayload {
|
||||
/// The input for a send: a cached file id goes out as `InputFile::file_id`
|
||||
/// (no fetch, no upload), URLs go to Telegram, anything else is a local
|
||||
/// path (transient upload fallback).
|
||||
fn input_file(&self) -> Result<InputFile, String> {
|
||||
match self {
|
||||
MediaItemPayload::Photo {
|
||||
media,
|
||||
file_id: true,
|
||||
..
|
||||
}
|
||||
| MediaItemPayload::Video {
|
||||
media,
|
||||
file_id: true,
|
||||
..
|
||||
}
|
||||
| MediaItemPayload::Animation {
|
||||
media,
|
||||
file_id: true,
|
||||
..
|
||||
} => Ok(InputFile::file_id(media.clone().into())),
|
||||
_ => input_file_for(item_url(self)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn photo_media(file: InputFile, caption: Option<&str>, spoiler: bool) -> InputMedia {
|
||||
let mut photo = InputMediaPhoto::new(file).parse_mode(ParseMode::Html);
|
||||
if let Some(caption) = caption {
|
||||
@@ -214,24 +391,22 @@ fn build_media_group(
|
||||
let item_caption = if i == 0 { caption } else { None };
|
||||
Ok(match item {
|
||||
MediaItemPayload::Photo {
|
||||
media,
|
||||
has_spoiler,
|
||||
} => photo_media(input_file_for(media)?, item_caption, *has_spoiler),
|
||||
has_spoiler, ..
|
||||
} => photo_media(item.input_file()?, item_caption, *has_spoiler),
|
||||
MediaItemPayload::Video {
|
||||
media,
|
||||
has_spoiler,
|
||||
thumbnail,
|
||||
..
|
||||
} => {
|
||||
let mut video = video_media(input_file_for(media)?, item_caption, *has_spoiler);
|
||||
let mut video = video_media(item.input_file()?, item_caption, *has_spoiler);
|
||||
if let (Some(thumb), InputMedia::Video(v)) = (thumbnail, &mut video) {
|
||||
*v = v.clone().thumbnail(input_file_for(thumb)?);
|
||||
}
|
||||
video
|
||||
}
|
||||
MediaItemPayload::Animation {
|
||||
media,
|
||||
has_spoiler,
|
||||
} => animation_media(input_file_for(media)?, item_caption, *has_spoiler),
|
||||
has_spoiler, ..
|
||||
} => animation_media(item.input_file()?, item_caption, *has_spoiler),
|
||||
})
|
||||
})
|
||||
.collect()
|
||||
@@ -258,8 +433,18 @@ fn sniff_ext(bytes: &[u8]) -> &'static str {
|
||||
enum FallbackError {
|
||||
Retryable { delay_seconds: f64 },
|
||||
Permanent { message: String },
|
||||
/// The downloaded file exceeds the upload cap; the caller falls back to
|
||||
/// the item's smaller URL.
|
||||
MediaTooLarge,
|
||||
}
|
||||
|
||||
/// Brings a downloaded photo within Telegram's limits via the pure-Rust
|
||||
/// chain in [`crate::photo`] (no ffmpeg): dimension cap / upload cap
|
||||
/// exceeded photos are decoded, downscaled with Lanczos3, PNG bit depth
|
||||
/// reduced (>24-bit → 24-bit RGB, ≤24-bit untouched) and transcoded to JPEG
|
||||
/// only if still too big. Anything that cannot be fixed falls back to the
|
||||
/// item's smaller URL.
|
||||
///
|
||||
/// Downloads one media item to a temp file (deleted on drop). Network errors
|
||||
/// are retryable; size over the upload cap and other download errors are not.
|
||||
async fn download_to_temp(item: &MediaItemPayload) -> Result<NamedTempFile, FallbackError> {
|
||||
@@ -281,10 +466,12 @@ async fn download_to_temp(item: &MediaItemPayload) -> Result<NamedTempFile, Fall
|
||||
});
|
||||
}
|
||||
};
|
||||
if bytes.len() as u64 > MAX_UPLOAD_BYTES {
|
||||
return Err(FallbackError::Permanent {
|
||||
message: "media too large".into(),
|
||||
});
|
||||
// Photos are downloaded even over the cap so `prepare_photo` can
|
||||
// downscale / transcode them; only videos/animations short-circuit.
|
||||
if !matches!(item, MediaItemPayload::Photo { .. })
|
||||
&& bytes.len() as u64 > MAX_UPLOAD_BYTES
|
||||
{
|
||||
return Err(FallbackError::MediaTooLarge);
|
||||
}
|
||||
let ext = sniff_ext(&bytes);
|
||||
let mut file = tempfile::Builder::new()
|
||||
@@ -302,7 +489,48 @@ async fn download_to_temp(item: &MediaItemPayload) -> Result<NamedTempFile, Fall
|
||||
Ok(file)
|
||||
}
|
||||
|
||||
/// Download-and-reupload fallback for one media batch.
|
||||
/// Builds the media group item from an uploaded file.
|
||||
fn media_from_file(
|
||||
item: &MediaItemPayload,
|
||||
path: std::path::PathBuf,
|
||||
caption: Option<&str>,
|
||||
) -> InputMedia {
|
||||
match item {
|
||||
MediaItemPayload::Photo { has_spoiler, .. } => {
|
||||
photo_media(InputFile::file(path), caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Video { has_spoiler, .. } => {
|
||||
video_media(InputFile::file(path), caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Animation { has_spoiler, .. } => {
|
||||
animation_media(InputFile::file(path), caption, *has_spoiler)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Builds the media group item from a (smaller) URL.
|
||||
fn media_from_url(
|
||||
item: &MediaItemPayload,
|
||||
url: &str,
|
||||
caption: Option<&str>,
|
||||
) -> Result<InputMedia, String> {
|
||||
Ok(match item {
|
||||
MediaItemPayload::Photo { has_spoiler, .. } => {
|
||||
photo_media(input_file_for(url)?, caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Video { has_spoiler, .. } => {
|
||||
video_media(input_file_for(url)?, caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Animation { has_spoiler, .. } => {
|
||||
animation_media(input_file_for(url)?, caption, *has_spoiler)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// Download-and-reupload fallback for one media batch. Files over the upload
|
||||
/// cap are not downloaded/uploaded; the item falls back to its smaller URL
|
||||
/// (which Telegram fetches itself). Returns the fallback-error without the
|
||||
/// task attached; callers wrap it with the updated task state.
|
||||
async fn send_batch_via_upload(
|
||||
bot: &Bot,
|
||||
chat_id: i64,
|
||||
@@ -313,22 +541,90 @@ async fn send_batch_via_upload(
|
||||
let mut files = Vec::new();
|
||||
let mut items = Vec::new();
|
||||
for (i, item) in batch.iter().enumerate() {
|
||||
let file = download_to_temp(item).await?;
|
||||
let path = file.path().to_path_buf();
|
||||
let item_caption = if i == 0 { caption } else { None };
|
||||
let media = match item {
|
||||
MediaItemPayload::Photo { has_spoiler, .. } => {
|
||||
photo_media(InputFile::file(path), item_caption, *has_spoiler)
|
||||
// Size check before downloading/uploading: over the cap, use the
|
||||
// smaller URL instead of the file. Photos are exempt — they are
|
||||
// downloaded and processed (downscale / PNG→JPEG) before uploading.
|
||||
let too_large = match x_media::site::media_size(item_url(item)).await {
|
||||
Ok(Some(size)) => size > MAX_UPLOAD_BYTES,
|
||||
_ => false,
|
||||
};
|
||||
let too_large = too_large && !matches!(item, MediaItemPayload::Photo { .. });
|
||||
let media = if too_large {
|
||||
match item.fallback_url() {
|
||||
Some(url) => match media_from_url(item, url, item_caption) {
|
||||
Ok(media) => media,
|
||||
Err(message) => {
|
||||
return Err(FallbackError::Permanent { message });
|
||||
}
|
||||
},
|
||||
None => {
|
||||
return Err(FallbackError::Permanent {
|
||||
message: "media too large".into(),
|
||||
});
|
||||
}
|
||||
}
|
||||
MediaItemPayload::Video { has_spoiler, .. } => {
|
||||
video_media(InputFile::file(path), item_caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Animation { has_spoiler, .. } => {
|
||||
animation_media(InputFile::file(path), item_caption, *has_spoiler)
|
||||
} else {
|
||||
match download_to_temp(item).await {
|
||||
Ok(file) => {
|
||||
// Telegram rejects photos wider+taller than 10000 px
|
||||
// combined (PHOTO_INVALID_DIMENSIONS): downscale the
|
||||
// downloaded file before uploading; photos that cannot be
|
||||
// brought within the limits degrade to the smaller URL.
|
||||
if matches!(item, MediaItemPayload::Photo { .. }) {
|
||||
// CPU-heavy (decode/resize/encode): run off the async
|
||||
// executor thread.
|
||||
let prep = tokio::task::spawn_blocking(move || photo::prepare_photo(file))
|
||||
.await
|
||||
.map_err(|e| FallbackError::Permanent {
|
||||
message: format!("photo worker panicked: {e}"),
|
||||
})?
|
||||
.map_err(|message| FallbackError::Permanent { message })?;
|
||||
match prep {
|
||||
PhotoPrep::Upload(upload) => {
|
||||
let path = upload.path().to_path_buf();
|
||||
files.push(upload);
|
||||
media_from_file(item, path, item_caption)
|
||||
}
|
||||
PhotoPrep::UseFallback => match item.fallback_url() {
|
||||
Some(url) => match media_from_url(item, url, item_caption) {
|
||||
Ok(media) => media,
|
||||
Err(message) => {
|
||||
return Err(FallbackError::Permanent { message });
|
||||
}
|
||||
},
|
||||
None => {
|
||||
return Err(FallbackError::Permanent {
|
||||
message:
|
||||
"photo dimensions exceed Telegram limits and no smaller variant is available"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
},
|
||||
}
|
||||
} else {
|
||||
let path = file.path().to_path_buf();
|
||||
files.push(file);
|
||||
media_from_file(item, path, item_caption)
|
||||
}
|
||||
}
|
||||
Err(FallbackError::MediaTooLarge) => match item.fallback_url() {
|
||||
Some(url) => match media_from_url(item, url, item_caption) {
|
||||
Ok(media) => media,
|
||||
Err(message) => {
|
||||
return Err(FallbackError::Permanent { message });
|
||||
}
|
||||
},
|
||||
None => {
|
||||
return Err(FallbackError::Permanent {
|
||||
message: "media too large".into(),
|
||||
});
|
||||
}
|
||||
},
|
||||
Err(e) => return Err(e),
|
||||
}
|
||||
};
|
||||
items.push(media);
|
||||
files.push(file);
|
||||
}
|
||||
let result = bot
|
||||
.send_media_group(ChatId(chat_id), items)
|
||||
@@ -355,12 +651,14 @@ fn updated_sequence_task(task: &Task, batch_index: usize, sent_message_ids: Vec<
|
||||
reply_to_message_id,
|
||||
caption,
|
||||
media_batches,
|
||||
batch_index: _,
|
||||
sent_message_ids: _,
|
||||
source_url,
|
||||
edit_before_forward,
|
||||
forward_channel_id,
|
||||
notify_chat_id,
|
||||
notify_message_id,
|
||||
..
|
||||
cache_data,
|
||||
} => Task::SendMediaSequence {
|
||||
chat_id: *chat_id,
|
||||
reply_to_message_id: *reply_to_message_id,
|
||||
@@ -373,6 +671,7 @@ fn updated_sequence_task(task: &Task, batch_index: usize, sent_message_ids: Vec<
|
||||
forward_channel_id: *forward_channel_id,
|
||||
notify_chat_id: *notify_chat_id,
|
||||
notify_message_id: *notify_message_id,
|
||||
cache_data: cache_data.clone(),
|
||||
},
|
||||
_ => unreachable!("updated_sequence_task requires a SendMediaSequence task"),
|
||||
}
|
||||
@@ -397,6 +696,10 @@ pub async fn send_media_sequence(bot: &Bot, task: &Task) -> Result<Vec<i64>, Sen
|
||||
let chat_id = *chat_id;
|
||||
let reply_to = *reply_to_message_id;
|
||||
let mut sent = sent_message_ids.clone();
|
||||
// File ids accumulated across batches for the link cache. Only a fresh
|
||||
// (non-resumed) full send populates the cache.
|
||||
let mut cached_media: Vec<CachedMedia> = Vec::new();
|
||||
let fresh_send = *batch_index == 0 && sent.is_empty();
|
||||
for idx in *batch_index..media_batches.len() {
|
||||
let batch = &media_batches[idx];
|
||||
let caption = if idx == 0 { Some(caption.as_str()) } else { None };
|
||||
@@ -420,18 +723,21 @@ pub async fn send_media_sequence(bot: &Bot, task: &Task) -> Result<Vec<i64>, Sen
|
||||
media_batches.len(),
|
||||
batch.len()
|
||||
);
|
||||
collect_file_ids(&messages, batch, &mut cached_media);
|
||||
sent.extend(messages.into_iter().map(|m| m.id.0 as i64));
|
||||
}
|
||||
Err(RequestError::Api(api)) if is_media_fetch_failure(&api) => {
|
||||
Err(RequestError::Api(api))
|
||||
if is_media_fetch_failure(&api) || is_size_error(&api) =>
|
||||
{
|
||||
log::info!(
|
||||
"Telegram could not fetch media for batch {idx} ({}), downloading and reuploading",
|
||||
batch
|
||||
.first()
|
||||
.map(|item| item_url(item))
|
||||
.unwrap_or("?")
|
||||
batch.first().map(item_url).unwrap_or("?")
|
||||
);
|
||||
match send_batch_via_upload(bot, chat_id, reply_to, batch, caption).await {
|
||||
Ok(messages) => sent.extend(messages.into_iter().map(|m| m.id.0 as i64)),
|
||||
Ok(messages) => {
|
||||
collect_file_ids(&messages, batch, &mut cached_media);
|
||||
sent.extend(messages.into_iter().map(|m| m.id.0 as i64));
|
||||
}
|
||||
Err(FallbackError::Retryable { delay_seconds }) => {
|
||||
return Err(SendError::Retryable {
|
||||
delay_seconds,
|
||||
@@ -444,6 +750,7 @@ pub async fn send_media_sequence(bot: &Bot, task: &Task) -> Result<Vec<i64>, Sen
|
||||
task: updated_sequence_task(task, idx, sent),
|
||||
});
|
||||
}
|
||||
Err(FallbackError::MediaTooLarge) => unreachable!("handled inside upload"),
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
@@ -454,6 +761,9 @@ pub async fn send_media_sequence(bot: &Bot, task: &Task) -> Result<Vec<i64>, Sen
|
||||
}
|
||||
}
|
||||
}
|
||||
if fresh_send {
|
||||
cache_sent_task(task, cached_media).await;
|
||||
}
|
||||
Ok(sent)
|
||||
}
|
||||
|
||||
@@ -494,6 +804,7 @@ pub async fn send_animation(bot: &Bot, task: &Task) -> Result<Vec<i64>, SendErro
|
||||
MediaItemPayload::Animation {
|
||||
media,
|
||||
has_spoiler,
|
||||
..
|
||||
} => (media, *has_spoiler),
|
||||
MediaItemPayload::Photo { .. } | MediaItemPayload::Video { .. } => {
|
||||
unreachable!("SendAnimation carries an Animation payload")
|
||||
@@ -506,34 +817,76 @@ pub async fn send_animation(bot: &Bot, task: &Task) -> Result<Vec<i64>, SendErro
|
||||
match send_animation_inner(bot, chat_id, reply_to, caption, has_spoiler, url_file)
|
||||
.await
|
||||
{
|
||||
Ok(message) => Ok(vec![message.id.0 as i64]),
|
||||
Err(RequestError::Api(api)) if is_media_fetch_failure(&api) => {
|
||||
Ok(message) => {
|
||||
let id = message.id.0 as i64;
|
||||
cache_animation_send(task, &message).await;
|
||||
Ok(vec![id])
|
||||
}
|
||||
Err(RequestError::Api(api))
|
||||
if is_media_fetch_failure(&api) || is_size_error(&api) =>
|
||||
{
|
||||
log::info!(
|
||||
"Telegram could not fetch animation URL, downloading and reuploading: {}",
|
||||
media_url
|
||||
);
|
||||
let file = match download_to_temp(animation).await {
|
||||
Ok(file) => file,
|
||||
match download_to_temp(animation).await {
|
||||
Ok(file) => {
|
||||
let path = file.path().to_path_buf();
|
||||
match send_animation_inner(
|
||||
bot,
|
||||
chat_id,
|
||||
reply_to,
|
||||
caption,
|
||||
has_spoiler,
|
||||
InputFile::file(path),
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(message) => {
|
||||
let id = message.id.0 as i64;
|
||||
cache_animation_send(task, &message).await;
|
||||
Ok(vec![id])
|
||||
}
|
||||
Err(e) => Err(classify_to_send_error(&e, task.clone())),
|
||||
}
|
||||
}
|
||||
// Over the upload cap: fall back to the smaller URL.
|
||||
Err(FallbackError::MediaTooLarge) => match animation.fallback_url() {
|
||||
Some(url) => match input_file_for(url) {
|
||||
Ok(file) => {
|
||||
match send_animation_inner(
|
||||
bot,
|
||||
chat_id,
|
||||
reply_to,
|
||||
caption,
|
||||
has_spoiler,
|
||||
file,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(message) => {
|
||||
let id = message.id.0 as i64;
|
||||
cache_animation_send(task, &message).await;
|
||||
Ok(vec![id])
|
||||
}
|
||||
Err(e) => Err(classify_to_send_error(&e, task.clone())),
|
||||
}
|
||||
}
|
||||
Err(message) => {
|
||||
Err(SendError::Permanent { message, task: task.clone() })
|
||||
}
|
||||
},
|
||||
None => Err(SendError::Permanent {
|
||||
message: "media too large".into(),
|
||||
task: task.clone(),
|
||||
}),
|
||||
},
|
||||
Err(FallbackError::Retryable { delay_seconds }) => {
|
||||
return Err(SendError::Retryable { delay_seconds, task: task.clone() });
|
||||
Err(SendError::Retryable { delay_seconds, task: task.clone() })
|
||||
}
|
||||
Err(FallbackError::Permanent { message }) => {
|
||||
return Err(SendError::Permanent { message, task: task.clone() });
|
||||
Err(SendError::Permanent { message, task: task.clone() })
|
||||
}
|
||||
};
|
||||
let path = file.path().to_path_buf();
|
||||
match send_animation_inner(
|
||||
bot,
|
||||
chat_id,
|
||||
reply_to,
|
||||
caption,
|
||||
has_spoiler,
|
||||
InputFile::file(path),
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(message) => Ok(vec![message.id.0 as i64]),
|
||||
Err(e) => Err(classify_to_send_error(&e, task.clone())),
|
||||
}
|
||||
}
|
||||
Err(e) => Err(classify_to_send_error(&e, task.clone())),
|
||||
@@ -731,6 +1084,7 @@ pub async fn handle_task(payload: serde_json::Value) -> Result<(), QueueError> {
|
||||
});
|
||||
}
|
||||
Err(SendError::Permanent { message, task }) => {
|
||||
invalidate_cache(&task).await;
|
||||
return Err(QueueError::Permanent {
|
||||
message,
|
||||
payload: serde_json::to_value(task).expect("task serializes"),
|
||||
@@ -782,6 +1136,14 @@ pub async fn dead_letter_notify(payload: serde_json::Value, message: String) {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn oversized_photo_boundary() {
|
||||
// The empirical Telegram limit: sum 10000 passes, 10001 fails.
|
||||
assert!(crate::photo::PHOTO_MAX_DIMENSION_SUM == 10000);
|
||||
assert!(6100 + 3900 <= crate::photo::PHOTO_MAX_DIMENSION_SUM);
|
||||
assert!(6300 + 3730 > crate::photo::PHOTO_MAX_DIMENSION_SUM);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chunk_media_items_sizes() {
|
||||
assert_eq!(chunk_media_items::<i32>(vec![]), Vec::<Vec<i32>>::new());
|
||||
@@ -819,6 +1181,36 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_size_error_matches_known_errors() {
|
||||
// 413 upload cap.
|
||||
let e = ApiError::RequestEntityTooLarge;
|
||||
assert!(is_size_error(&e), "{e:?}");
|
||||
// Unknown descriptions with size wording.
|
||||
for description in [
|
||||
"Bad Request: file is too large",
|
||||
"Bad Request: media is too big",
|
||||
"Bad Request: url file size is too big",
|
||||
] {
|
||||
let api = ApiError::Unknown(description.to_string());
|
||||
assert!(is_size_error(&api), "{description}");
|
||||
}
|
||||
// Unrelated errors must not match.
|
||||
for description in ["Bad Request: WEBPAGE_MEDIA_EMPTY", "Bad Request: message is not modified"] {
|
||||
let api = ApiError::Unknown(description.to_string());
|
||||
assert!(!is_size_error(&api), "{description}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn media_item_payload_fallback_url_serde_default() {
|
||||
// Old queued payloads without the field deserialize with None.
|
||||
let json = serde_json::json!({"kind": "photo", "media": "https://a/b.jpg", "has_spoiler": false});
|
||||
let photo: MediaItemPayload = serde_json::from_value(json).unwrap();
|
||||
assert!(matches!(photo, MediaItemPayload::Photo { fallback_url: None, .. }));
|
||||
assert_eq!(photo.fallback_url(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn classification_mapping() {
|
||||
use teloxide::types::Seconds;
|
||||
@@ -858,11 +1250,15 @@ mod tests {
|
||||
vec![MediaItemPayload::Photo {
|
||||
media: "https://a/b.jpg".into(),
|
||||
has_spoiler: true,
|
||||
fallback_url: Some("https://a/b_small.jpg".into()),
|
||||
file_id: false,
|
||||
}],
|
||||
vec![MediaItemPayload::Video {
|
||||
media: "https://a/v.mp4".into(),
|
||||
has_spoiler: false,
|
||||
thumbnail: Some("https://a/t.jpg".into()),
|
||||
fallback_url: None,
|
||||
file_id: false,
|
||||
}],
|
||||
],
|
||||
batch_index: 1,
|
||||
@@ -872,6 +1268,7 @@ mod tests {
|
||||
forward_channel_id: Some(333),
|
||||
notify_chat_id: Some(111),
|
||||
notify_message_id: Some(222),
|
||||
cache_data: None,
|
||||
};
|
||||
let json = serde_json::to_value(&task).unwrap();
|
||||
assert_eq!(json["type"], "send_media_sequence");
|
||||
@@ -900,6 +1297,8 @@ mod tests {
|
||||
let photo = MediaItemPayload::Photo {
|
||||
media: "https://a/b.jpg".into(),
|
||||
has_spoiler: false,
|
||||
fallback_url: None,
|
||||
file_id: false,
|
||||
};
|
||||
let json = serde_json::to_value(&photo).unwrap();
|
||||
assert_eq!(json["kind"], "photo");
|
||||
|
||||
@@ -74,6 +74,10 @@ impl ChatStore {
|
||||
let db_path = self.db_path.clone();
|
||||
let payload = tokio::task::spawn_blocking(move || -> rusqlite::Result<Option<String>> {
|
||||
let conn = Connection::open(&db_path)?;
|
||||
// Concurrent handler tasks (batch-forwards) may write chat_state
|
||||
// while this read runs; without a busy timeout a write lock
|
||||
// collision fails the query immediately.
|
||||
conn.busy_timeout(std::time::Duration::from_secs(5))?;
|
||||
let mut stmt = conn.prepare("SELECT payload FROM chat_state WHERE chat_id = ?1")?;
|
||||
let mut rows = stmt.query(params![chat_id.to_string()])?;
|
||||
match rows.next()? {
|
||||
@@ -100,6 +104,7 @@ impl ChatStore {
|
||||
let db_path = self.db_path.clone();
|
||||
tokio::task::spawn_blocking(move || -> rusqlite::Result<()> {
|
||||
let conn = Connection::open(&db_path)?;
|
||||
conn.busy_timeout(std::time::Duration::from_secs(5))?;
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO chat_state (chat_id, payload) VALUES (?1, ?2)",
|
||||
params![chat_id.to_string(), payload],
|
||||
|
||||
+50
-14
@@ -1,30 +1,66 @@
|
||||
services:
|
||||
nginx-proxy:
|
||||
image: nginxproxy/nginx-proxy:1.11.6-alpine
|
||||
restart: always
|
||||
ports:
|
||||
- '80:80'
|
||||
- '443:443'
|
||||
volumes:
|
||||
- /var/run/docker.sock:/tmp/docker.sock:ro
|
||||
- certs:/etc/nginx/certs:ro
|
||||
- html:/usr/share/nginx/html:ro
|
||||
networks: [proxy]
|
||||
labels:
|
||||
- 'com.github.nginx-proxy.nginx=true'
|
||||
container_name: nginx-proxy
|
||||
|
||||
acme-companion:
|
||||
image: nginxproxy/acme-companion
|
||||
restart: always
|
||||
environment:
|
||||
DEFAULT_EMAIL: 'admin@yoursfunny.top'
|
||||
volumes:
|
||||
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||
- certs:/etc/nginx/certs:rw
|
||||
- html:/usr/share/nginx/html:rw
|
||||
- acme:/etc/acme.sh
|
||||
networks: [proxy]
|
||||
container_name: acme-companion
|
||||
depends_on:
|
||||
- nginx-proxy
|
||||
|
||||
tgxmb:
|
||||
image: yoursfunny/telegram-twitter-media-bot:latest
|
||||
restart: always
|
||||
# ports:
|
||||
# - "8443:8443"
|
||||
environment:
|
||||
# docker-entrypoint.sh drops privileges to this uid.
|
||||
LOCAL_USER_ID: '1000'
|
||||
# Bot token (BotFather). Required.
|
||||
TELOXIDE_TOKEN: ''
|
||||
# Comma-separated admin chat ids; receives startup/shutdown notices.
|
||||
BOT_ADMIN: ''
|
||||
# Required for pixiv support; pixiv is disabled when unset.
|
||||
PIXIV_REFRESH_TOKEN: ''
|
||||
# Edit-before-forward records expire after this many seconds (default 86400 = 24h).
|
||||
TWITTER_AUTH_TOKEN: ''
|
||||
EDIT_MESSAGE_TTL_SECONDS: '86400'
|
||||
LINK_CACHE_TTL_SECONDS: '604800'
|
||||
RUST_LOG: 'info'
|
||||
# Webhook mode is off by default (polling). The listener binds inside the
|
||||
# container, so use 0.0.0.0 and publish the port if you enable it.
|
||||
WEBHOOK: 'false'
|
||||
VIRTUAL_HOST: '<YOUR_DOMAIN>'
|
||||
VIRTUAL_PORT: '8443'
|
||||
# ACME_HOST: 'your.domain.com'
|
||||
WEBHOOK: 'true'
|
||||
WEBHOOK_LISTEN: '0.0.0.0'
|
||||
WEBHOOK_PORT: '8443'
|
||||
WEBHOOK_URL: 'https://example.com'
|
||||
WEBHOOK_CERT: './cert/cert.pem'
|
||||
WEBHOOK_SECRET_TOKEN: 'secret-token'
|
||||
WEBHOOK_URL: 'https://<YOUR_DOMAIN>/'
|
||||
WEBHOOK_SECRET_TOKEN: ''
|
||||
volumes:
|
||||
- ./data:/app/data
|
||||
# - ./cert:/app/cert
|
||||
networks: [proxy]
|
||||
depends_on:
|
||||
- nginx-proxy
|
||||
container_name: tgxmb
|
||||
|
||||
volumes:
|
||||
certs:
|
||||
html:
|
||||
acme:
|
||||
|
||||
networks:
|
||||
proxy:
|
||||
name: proxy
|
||||
|
||||
+15
-3
@@ -5,9 +5,21 @@ if [ "$(id -u)" -eq '0' ]
|
||||
then
|
||||
USER_ID=${LOCAL_USER_ID:-9001}
|
||||
|
||||
useradd --shell /bin/bash -u ${USER_ID} -o -c "" -m user > /dev/null 2>&1
|
||||
usermod -a -G root user > /dev/null 2>&1
|
||||
chown -R `id -u user`:`id -u user` /app > /dev/null 2>&1
|
||||
# `docker compose restart` / `docker restart` reuse the same container, so
|
||||
# the overlay fs keeps the user created on first boot. A second `useradd`
|
||||
# then fails with exit code 9, which would trip `set -e` and kill the
|
||||
# container on every restart. Create only if missing; align the UID
|
||||
# otherwise so LOCAL_USER_ID changes still apply.
|
||||
if ! id user > /dev/null 2>&1
|
||||
then
|
||||
useradd --shell /bin/bash -u ${USER_ID} -o -c "" -m user > /dev/null 2>&1 || true
|
||||
else
|
||||
usermod -u ${USER_ID} -o user > /dev/null 2>&1 || true
|
||||
fi
|
||||
usermod -a -G root user > /dev/null 2>&1 || true
|
||||
# Bind-mounted volumes may not support chown; a failure here must not kill
|
||||
# the container either.
|
||||
chown -R `id -u user`:`id -u user` /app > /dev/null 2>&1 || true
|
||||
|
||||
export HOME=/home/user
|
||||
# setpriv (util-linux, present in bookworm-slim) replaces gosu: drop to the
|
||||
|
||||
Reference in New Issue
Block a user