mirror of
https://github.com/TheFunny/TelegramTwitterMediaBot.git
synced 2026-09-23 23:32:05 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fb601f4d5d
|
||
|
|
d8dd4fa91e
|
||
|
|
af96caff40
|
||
|
|
52184ba6fb
|
||
|
|
0eb4e5c78d
|
||
|
|
5c51de217a
|
||
|
|
c1f5d3ca54
|
||
|
|
1bb6968108 | ||
|
|
14b444d109
|
||
|
|
de3105d4cd | ||
|
|
c46103a23f | ||
|
|
9529f64b41
|
||
|
|
d0de17329d | ||
|
|
9af37e92b4 | ||
|
|
e68a1dbd30 | ||
|
|
b9c6d16ff0
|
||
|
|
ec65c3ce74
|
||
|
|
893ab7a1e0
|
||
|
|
bd032e3d68
|
||
|
|
dbda6ec1c2
|
||
|
|
0a9ff58a69
|
||
|
|
c2d7c8406e
|
||
|
|
abdc27ed5e
|
||
|
|
3fb8421c3a
|
||
|
|
475cfd18f9
|
||
|
|
0a577600fd
|
||
|
|
32254fa807
|
||
|
|
11c04b66dc
|
||
|
|
2f741e5f4b
|
||
|
|
89c4642e1c
|
||
|
|
3f6a0f034a
|
||
|
|
90a011e978
|
||
|
|
12a065846c
|
||
|
|
dca1eff1c9
|
||
|
|
894a9ebf4a
|
||
|
|
c968891ff6
|
||
|
|
0087bd01ac
|
||
|
|
4cb40909c5
|
||
|
|
f260f41755
|
||
|
|
af901caddb
|
||
|
|
c0af42b1cc
|
||
|
|
d0810217b4
|
||
|
|
e6ba178983
|
||
|
|
ae69d72930
|
||
|
|
1e30815a10
|
||
|
|
50206a9056
|
||
|
|
c9e72fda70
|
||
|
|
f6845b1b5c
|
||
|
|
fae8dc6f2d
|
||
|
|
ac72e414c3
|
||
|
|
8b3b2a246b
|
||
|
|
6f6898c245
|
||
|
|
69698992d5
|
||
|
|
1e77bb0478
|
||
|
|
8f2b0a1dcb
|
||
|
|
b65fb967c4
|
||
|
|
a8fd685777
|
||
|
|
5679a8c172
|
||
|
|
bf4e6159b3
|
||
|
|
5e23916b40
|
||
|
|
7ca8fd1da2
|
||
|
|
96c11becb9
|
||
|
|
5830a3f013
|
||
|
|
183bb7e435
|
||
|
|
2a8433a8d2
|
||
|
|
6b3e61881d
|
||
|
|
47935dd7c6
|
||
|
|
6911e9146e
|
||
|
|
aa705aef90
|
||
|
|
39260a8817
|
||
|
|
6b9640aa48
|
||
|
|
ad59f518ff
|
||
|
|
e21643063e
|
||
|
|
246fc989f0
|
||
|
|
2e2d1b3506
|
||
|
|
1747d321d8
|
||
|
|
505990e49e
|
||
|
|
95b475ff08
|
||
|
|
2297fdc91c
|
||
|
|
4580b79d4f
|
||
|
|
edb32c23b4
|
||
|
|
4a467641aa
|
||
|
|
bd43a12dee
|
||
|
|
c496e41c55
|
||
|
|
68f026c990
|
||
|
|
62d80c8905
|
||
|
|
78c9c841c6
|
||
|
|
e49d500d23
|
||
|
|
6feabd723b
|
||
|
|
ea72516d5c
|
||
|
|
755330e585
|
||
|
|
c40b074b3c
|
||
|
|
9910da2914
|
||
|
|
ebc0122264
|
||
|
|
44cba8abe0
|
||
|
|
99009aae9a
|
||
|
|
72130b9023
|
||
|
|
734cfc2eb3
|
||
|
|
a8156697fa
|
||
|
|
c093dfe5ac
|
||
|
|
ee6f3e4a27
|
||
|
|
16ed53fead
|
||
|
|
1d9e3629c9
|
||
|
|
b3d87b4f7d
|
||
|
|
98c48b99c0
|
||
|
|
f40639c799
|
||
|
|
51cc079a85
|
||
|
|
aa3083792a
|
||
|
|
6849006ad7
|
||
|
|
042a04ab6e
|
||
|
|
9f28af4e6b
|
||
|
|
d61dba5096
|
||
|
|
f6df3e28cb
|
||
|
|
deb1ef2428
|
||
|
|
b5e5340edc
|
||
|
|
425d1505cf
|
||
|
|
6e40f55440
|
||
|
|
7998114dc3
|
||
|
|
9a96f78177
|
||
|
|
b50f794d52
|
||
|
|
d32fa969d6
|
@@ -26,6 +26,10 @@
|
||||
*.db
|
||||
LICENSE
|
||||
README.md
|
||||
# Documentation and scratch files: the build only ever reads the manifests,
|
||||
# `crates/` and the entrypoint script.
|
||||
docs/
|
||||
*.md
|
||||
data/
|
||||
cert/
|
||||
nginx-certs/
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
version: 2
|
||||
|
||||
# Pairs with the `actions-rust-lang/audit` gate in ci.yml: the gate reports
|
||||
# advisories in Cargo.lock, this is what actually moves the dependencies.
|
||||
# Patch bumps are batched into one PR; minor/major stay separate so they get
|
||||
# reviewed and tested individually.
|
||||
updates:
|
||||
- package-ecosystem: cargo
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
open-pull-requests-limit: 5
|
||||
groups:
|
||||
cargo-patch:
|
||||
applies-to: version-updates
|
||||
patterns: ['*']
|
||||
update-types: ['patch']
|
||||
|
||||
# The workflow actions are pinned to commit SHAs; that pin is what makes
|
||||
# bumping them a manual chore, so let the bot do it.
|
||||
- package-ecosystem: github-actions
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
open-pull-requests-limit: 5
|
||||
|
||||
# The Dockerfile's base images (rust:1-bookworm, debian:bookworm-slim).
|
||||
- package-ecosystem: docker
|
||||
directory: /
|
||||
schedule:
|
||||
interval: monthly
|
||||
open-pull-requests-limit: 3
|
||||
@@ -0,0 +1,106 @@
|
||||
name: CI
|
||||
|
||||
# Test/lint gate (offline, no secrets) on every push/PR, plus a live-network
|
||||
# job that exercises the real source sites and the token-gated pixiv tests.
|
||||
#
|
||||
# Layering:
|
||||
# test — fmt + clippy + the full offline unit suite + a release-profile
|
||||
# build + cargo-audit dependency gate. Runs on every push and PR,
|
||||
# including forks (it needs no secrets).
|
||||
# live — the #[ignore]d live-network tests plus the pixiv tests that are
|
||||
# gated on PIXIV_REFRESH_TOKEN. Runs on schedule / manual dispatch
|
||||
# / tag pushes only, because pull requests from forks cannot read
|
||||
# repository secrets. continue-on-error keeps a flaky external site
|
||||
# from blocking, while the run still records the outcome.
|
||||
#
|
||||
# Every action is pinned to a commit SHA (Dependabot keeps the pins current);
|
||||
# `dtolnay/rust-toolchain` deliberately stays on its channel ref, because the
|
||||
# ref itself is what selects the toolchain (`@stable` = install stable).
|
||||
#
|
||||
# Test gating convention (keep in sync with AGENTS.md "Testing & QA"):
|
||||
# - pure unit tests: plain #[test] / #[tokio::test], always run.
|
||||
# - live-network tests: #[ignore = "live network: ..."], only run here.
|
||||
# - token-gated tests (pixiv): #[tokio::test] with an early return when
|
||||
# PIXIV_REFRESH_TOKEN is absent or empty (empty = unset CI secret).
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [master]
|
||||
pull_request:
|
||||
schedule:
|
||||
# Weekly probe of the live endpoints, so external API changes surface.
|
||||
- cron: '0 3 * * 1'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# A newer push to the same ref supersedes the older run; without this every
|
||||
# intermediate commit of a PR branch keeps a runner busy to completion.
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# Panicking tests print their backtrace; free when nothing fails.
|
||||
RUST_BACKTRACE: 1
|
||||
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
# Generous on purpose: the release-profile build below is cold on the very
|
||||
# first run (thin LTO + codegen-units = 1 across every dependency), and a
|
||||
# timeout there would kill the job *before* rust-cache saves its cache —
|
||||
# leaving every later run cold again.
|
||||
timeout-minutes: 45
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
components: clippy, rustfmt
|
||||
- uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2
|
||||
# `--locked` on every cargo invocation: the version bump edits
|
||||
# Cargo.lock by hand (AGENTS.md), so a stale lock must fail here instead
|
||||
# of being silently re-resolved — otherwise CI tests a different
|
||||
# dependency set than the one committed, and than the one the released
|
||||
# image is built from.
|
||||
- name: Check formatting
|
||||
run: cargo fmt --check
|
||||
- name: Lint (deny warnings)
|
||||
run: cargo clippy --workspace --all-targets --locked -- -D warnings
|
||||
- name: Run offline tests
|
||||
run: cargo test --workspace --locked
|
||||
# The release profile (lto/strip/codegen-units=1, overflow checks off)
|
||||
# was otherwise only exercised by the Docker build on master/tag. Same
|
||||
# package the Dockerfile builds; the cache keeps it cheap after the
|
||||
# first run.
|
||||
- name: Build release profile
|
||||
run: cargo build --release --locked -p xmedia-bot
|
||||
# Dependency vulnerability gate: fails the build when a crate in
|
||||
# Cargo.lock has an unfixed security advisory. Unmaintained/unsound
|
||||
# *warnings* (dotenv, proc-macro-error2, anyhow transitive) do not fail
|
||||
# the build by default; the advisory DB is cached across runs.
|
||||
- name: Audit dependencies
|
||||
uses: actions-rust-lang/audit@72c09e02f132669d52284a3323acdb503cfc1a24 # v1
|
||||
|
||||
live:
|
||||
needs: test
|
||||
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' || startsWith(github.ref, 'refs/tags/v')
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
continue-on-error: true
|
||||
env:
|
||||
PIXIV_REFRESH_TOKEN: ${{ secrets.PIXIV_REFRESH_TOKEN }}
|
||||
TWITTER_AUTH_TOKEN: ${{ secrets.TWITTER_AUTH_TOKEN }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: dtolnay/rust-toolchain@stable
|
||||
- uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2
|
||||
# Everything network- or secret-gated lives in x-media, and the bot
|
||||
# crate's suite (MockSender + tempdir stores, no network) already ran in
|
||||
# the `test` job — rebuilding it here bought nothing.
|
||||
- name: Run token-gated tests
|
||||
run: cargo test -p x-media --locked
|
||||
# The live-network tests, by the "live" name filter (all #[ignore]d).
|
||||
- name: Run live-network tests
|
||||
run: cargo test -p x-media --locked -- --ignored live
|
||||
@@ -1,28 +1,75 @@
|
||||
name: Build Docker Image
|
||||
|
||||
# Release builds (master / v* tags) plus a build-only check on pull requests
|
||||
# that touch anything the image depends on — the Dockerfile's stub-source
|
||||
# machinery, the ffmpeg download and the entrypoint are exactly the parts that
|
||||
# would otherwise break only at release time.
|
||||
#
|
||||
# Actions are pinned to commit SHAs (Dependabot keeps the pins current).
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- v*
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
paths:
|
||||
- Dockerfile
|
||||
- docker-entrypoint.sh
|
||||
- .dockerignore
|
||||
- Cargo.toml
|
||||
- Cargo.lock
|
||||
- .github/workflows/docker.yml
|
||||
- 'crates/**/Cargo.toml'
|
||||
|
||||
env:
|
||||
APP_NAME: telegram-twitter-media-bot
|
||||
DOCKERHUB_REPO: yoursfunny/telegram-twitter-media-bot
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Serialize runs per ref. Never cancel in progress: a killed run would drop a
|
||||
# half-finished image push.
|
||||
concurrency:
|
||||
group: docker-${{ github.ref }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
# A tag push and a branch push to the same commit fire two workflow runs;
|
||||
# build only once. Tag runs always build; master runs build only when the
|
||||
# pushed commit is not already tagged (the tag run covers it).
|
||||
should-build:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
build: ${{ steps.check.outputs.build }}
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
# A release tag is the version claim: the manifests are bumped by hand,
|
||||
# so `v1.5.1` with `Cargo.toml` still at 1.5.0 would publish an image
|
||||
# whose tag lies about what is inside it (the binary carries no version).
|
||||
- name: Verify the tag matches both crate versions
|
||||
if: startsWith(github.ref, 'refs/tags/v')
|
||||
shell: bash
|
||||
run: |
|
||||
tag="${GITHUB_REF_NAME#v}"
|
||||
status=0
|
||||
for manifest in crates/x-media/Cargo.toml crates/xmedia-bot/Cargo.toml; do
|
||||
# tr -d '\r': a CRLF checkout (core.autocrlf on Windows) would
|
||||
# otherwise yield "1.5.0\r" and false-fail every tag.
|
||||
version="$(sed -n 's/^version = "\(.*\)"/\1/p' "$manifest" | head -1 | tr -d '\r')"
|
||||
if [ "$version" != "$tag" ]; then
|
||||
echo "::error file=$manifest::$manifest is at $version but the tag is v$tag"
|
||||
status=1
|
||||
else
|
||||
echo "$manifest: $version matches v$tag"
|
||||
fi
|
||||
done
|
||||
exit "$status"
|
||||
- id: check
|
||||
shell: bash
|
||||
run: |
|
||||
@@ -33,14 +80,20 @@ jobs:
|
||||
echo "build=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# No `actions/checkout` here on purpose: `docker/build-push-action` defaults
|
||||
# to the Git context (`https://github.com/<owner>/<repo>.git#<ref>`), so
|
||||
# BuildKit clones the repo itself and authenticates with the automatic
|
||||
# github.token. Adding `context: .` below without a checkout step would hand
|
||||
# BuildKit an empty workspace.
|
||||
docker:
|
||||
needs: should-build
|
||||
if: needs.should-build.outputs.build == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: Docker meta
|
||||
id: meta
|
||||
uses: docker/metadata-action@v6
|
||||
uses: docker/metadata-action@dc802804100637a589fabce1cb79ff13a1411302 # v6
|
||||
with:
|
||||
images: ${{ env.DOCKERHUB_REPO }}
|
||||
tags: |
|
||||
@@ -49,15 +102,15 @@ jobs:
|
||||
type=semver,pattern={{major}}.{{minor}}
|
||||
type=semver,pattern={{major}}
|
||||
type=sha
|
||||
-
|
||||
name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v4
|
||||
-
|
||||
name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v4
|
||||
uses: docker/setup-buildx-action@f87e5991a6d7451dcb8d9637bfbc97413f497069 # v4
|
||||
# Pull requests build the image to prove the Dockerfile still works, but
|
||||
# must not read registry credentials (fork PRs have none).
|
||||
-
|
||||
name: Login to Docker Hub
|
||||
uses: docker/login-action@v4
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
@@ -66,15 +119,26 @@ jobs:
|
||||
# stage's layers so the cargo-deps and ffmpeg layers are restored
|
||||
# instead of re-downloaded/recompiled. The scope must be pinned to a
|
||||
# fixed string: the gha backend defaults to the current git ref, which
|
||||
# would give every new tag a cold cache on release builds.
|
||||
# would give every new tag a cold cache on release builds. PR runs only
|
||||
# read it (cache-to is empty) so they cannot evict the release cache.
|
||||
#
|
||||
# FFMPEG_URL/FFMPEG_SHA256 come from repository variables when set, so a
|
||||
# release can pin an exact ffmpeg build (the Dockerfile default follows
|
||||
# the project's `/redirect/latest/` URL, which has no sha256 sidecar).
|
||||
#
|
||||
# Single-arch (amd64) on purpose: adding arm64 means re-adding
|
||||
# `docker/setup-qemu-action`, `platforms: linux/amd64,linux/arm64`, and
|
||||
# parameterizing FFMPEG_URL by $TARGETARCH in the Dockerfile.
|
||||
-
|
||||
name: Build and push
|
||||
uses: docker/build-push-action@v7
|
||||
uses: docker/build-push-action@c3c9e263c25d99ce0380d002d59b67737d91b0dc # v7
|
||||
with:
|
||||
push: true
|
||||
push: ${{ github.event_name != 'pull_request' }}
|
||||
build-args: |
|
||||
APP_NAME=${{ env.APP_NAME }}
|
||||
FFMPEG_URL=${{ vars.FFMPEG_URL || 'https://ffmpeg.martin-riedl.de/redirect/latest/linux/amd64/release/ffmpeg.zip' }}
|
||||
FFMPEG_SHA256=${{ vars.FFMPEG_SHA256 }}
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
cache-from: type=gha,scope=tgxmb-build
|
||||
cache-to: type=gha,mode=max,scope=tgxmb-build
|
||||
cache-to: ${{ github.event_name != 'pull_request' && 'type=gha,mode=max,scope=tgxmb-build' || '' }}
|
||||
|
||||
@@ -2,11 +2,11 @@
|
||||
|
||||
## Project Overview
|
||||
|
||||
Telegram bot (teloxide) that turns post links from X/Twitter, Pixiv, and Bluesky into media messages (images, video, GIF) with the post's title, author, and tags. It supports batch media splitting, retry with persistence, inline queries, forward-channel rebinding with caption templates, and Pixiv ugoira→MP4 transcoding. README and user-facing strings are in Chinese. The project is a Rust port of a Python predecessor (see `queue.rs` comments referencing `utils/task_queue.py`).
|
||||
Telegram bot (teloxide) that turns post links from X/Twitter, Pixiv, Bluesky, Misskey (misskey.io), and Bilibili dynamics into media messages (images, video, GIF) with the post's title, author, and tags. It supports batch media splitting, retry with persistence, inline queries, forward-channel rebinding with caption templates, and Pixiv ugoira→MP4 transcoding. README is in Chinese; user-facing bot strings are in English. The project is a Rust port of a Python predecessor (see `queue.rs` comments referencing `utils/task_queue.py`).
|
||||
|
||||
Two-crate Cargo workspace (both v1.0.3, edition 2024, resolver 3):
|
||||
Two-crate Cargo workspace (both v1.7.0, edition 2024, resolver 3):
|
||||
|
||||
- **`crates/x-media`** — library that fetches and normalizes media from the three sites. Pure, no Telegram knowledge.
|
||||
- **`crates/x-media`** — library that fetches and normalizes media from the four sites. Pure, no Telegram knowledge.
|
||||
- **`crates/xmedia-bot`** — the bot binary: teloxide dispatcher, SQLite-backed chat state, persistent task queue.
|
||||
|
||||
## Architecture & Data Flow
|
||||
@@ -18,23 +18,31 @@ Telegram update → Dispatcher (polling or axum webhook) → dptree branches
|
||||
└─ callback_query → "forward" (copy to channel) / "template|<name>" (apply caption template)
|
||||
```
|
||||
|
||||
Message flow: `message_handler` extracts URLs (from `url`/`text_link` entities, text + caption, deduped) → `x_media::site::fetch(url)` → `Fetched` → builds a `Task` → `send::send_media_sequence` (media groups ≤ 9, caption on first item) or `send::send_animation`. On Telegram URL-fetch failure or size error (`send_batch_via_upload`): download via `x_media::site::download_media` to a temp file (≤ 10 MiB), sniff magic bytes (`sniff_ext`), upload via multipart; oversized items fall back to `fallback_url`. On failure: `enqueue_retry` persists resume-state `Task` into the SQLite queue → single worker leases (120 s lock TTL) → retry with exponential backoff (≤ 30 s, `MAX_RETRIES = 2`) → dead-letter → `notify_failure`. Success → `post_send_actions`: edit-before-forward prompt with inline buttons, or `copy_messages` to the bound forward channel.
|
||||
Message flow: `message_handler` extracts URLs (from `url`/`text_link` entities, text + caption, deduped) → `x_media::site::fetch(url)` → `Fetched` → builds a `Task` → `send::send_media_sequence` (media groups ≤ 9, caption on first item) or `send::send_animation`. On Telegram URL-fetch failure or size error (`send_batch_via_upload`): download via `x_media::site::download_media` to a temp file (≤ 10 MiB), sniff magic bytes (`sniff_ext`), upload via multipart; oversized items fall back to `fallback_url`. On failure: `enqueue_retry` persists resume-state `Task` into the SQLite queue → workers lease (120 s lock TTL) → retry with exponential backoff (≤ 30 s, `MAX_RETRIES = 2`) → dead-letter → `notify_failure`. Success → `post_send_actions`: edit-before-forward prompt with inline buttons, or `copy_messages` to the bound forward channel.
|
||||
|
||||
The `x-media` library: `site::fetch(url)` dispatches (in order) twitter → bsky → pixiv via per-site regex `PATTERN` and returns `Ok(None)` for unmatched URLs. `Fetched { source_url, caption, title, media: Vec<Media>, sensitive, … }`; `caption_with(format)` substitutes `{url} {author} {author_url} {title} {tags}`.
|
||||
Debug command: `/debug <url>` runs the same `x_media::site::fetch` and replies with `debug_report` (`handlers/commands.rs`) — site id, normalized cache key, source URL, title/author/tags, sensitive flag, caption and the media list — nothing is sent, cached or forwarded; the report is capped at 4000 chars and sent with HTML parse mode: raw fields are escaped, and the caption is wrapped in a `<blockquote>` so it renders exactly like the sent media caption (escaped text and links included).
|
||||
|
||||
The `/test <url>` command runs the ordinary link pipeline (`urls::url_media`) with `PostSend::Suppressed`: the media is sent and cached like any other link, but the chat's `forward_channel_id`/`edit_before_forward` are ignored, so a test never forwards to the channel and never opens the edit prompt (retries and dead-letter notifications behave as usual). Both commands use a custom `parse_arg_remainder` parser (whole remainder, trimmed) because teloxide's built-in `split` parser takes exactly one space-separated token.
|
||||
|
||||
The `x-media` library: `site::fetch(url)` dispatches through the `SITES` registry (per-site `impl Site`, in order twitter → bsky → misskey → pixiv → bilibili) and returns `Ok(None)` for unmatched URLs. `Fetched { source_url, caption, title, content, media: Vec<Media>, sensitive, site_id, … }` (title and content are split per platform: a pixiv artwork's title and description, a bilibili headline and body, and text-only posts whose text is all `content`); `caption_with(format)` substitutes `{url} {author} {author_url} {title} {content} {tags}`.
|
||||
|
||||
## Key Directories
|
||||
|
||||
| Path | Purpose |
|
||||
|---|---|
|
||||
| `crates/x-media/src/` | Fetch library. `site/mod.rs` = dispatcher + `Fetched`/`FetchError`/`download_media`/`media_size`; `media.rs` = `Media` enum; `examples/fetch.rs` = end-to-end usage sample |
|
||||
| `crates/x-media/src/site/<twitter\|pixiv\|bsky>/` | One directory per site: `mod.rs` (re-exports), `interface.rs` (PATTERN, `enabled()`, `fetch_from_url()`, site struct, `From<SiteStruct> for Fetched`), `model.rs` (serde DTOs). Pixiv adds `api.rs` (auth + transport); twitter adds `auth.rs` (logged-in GraphQL `TweetDetail` fallback for NSFW tweets, gated on `TWITTER_AUTH_TOKEN`) |
|
||||
| `crates/xmedia-bot/src/main.rs` | Entry point: env/log init, queue worker start, pixiv validation, 300 s edit-expiry sweep, dptree handler tree, webhook vs polling dispatch |
|
||||
| `crates/x-media/src/site/<twitter\|pixiv\|bsky\|misskey\|bilibili>/` | One directory per site: `mod.rs` (re-exports), `interface.rs` (PATTERN, `enabled()`, `fetch_from_url()`, `cache_key`/`is_retryable`/`media_headers`, unit struct `<Name>Site` implementing `site::Site`, `From<SiteStruct> for Fetched`), `model.rs` (serde DTOs). Pixiv adds `api.rs` (auth + transport); twitter adds `auth.rs` (logged-in GraphQL `TweetDetail` fallback for NSFW tweets, gated on `TWITTER_AUTH_TOKEN`). Misskey targets misskey.io only (`POST /api/notes/show`, 400+`NO_SUCH_NOTE` → NotFound). Bilibili fetches dynamics (images/animated images only — an attached video degrades to its cover, and its title stands in for the post text, which AV dynamics do not have) from `/x/polymer/web-dynamic/v1/detail` sent with `features=itemOpusStyle` (without that flag the legacy serialization drops an image/text post's body and headline entirely — `desc` comes back `null`; the adapter still parses the legacy `major.draw`/`desc`/`archive` shapes as a fallback). No WBI signature is involved; device cookies `buvid3`/`buvid4` are fetched automatically from `/x/frontend/finger/spi` because bilibili's `-352` risk control starts rejecting plain requests, `BILIBILI_COOKIE` is the escalation when an IP stays blocked; `b23.tv` short links are deliberately unmatched. Twitter's `from_syndication_json` HTML-decodes the API text — syndication and GraphQL `full_text` both arrive pre-escaped (`>` `<` `&` `'`) — so the stored text is raw and the caption escap…
|
||||
| `crates/xmedia-bot/src/main.rs` | Entry point: env/log init, command registration (`register_commands`), shared `send::BOT` force-init, queue worker start, site login validation (`site::validate_all`), 300 s edit-expiry sweep, dptree handler tree, webhook vs polling dispatch |
|
||||
| `crates/xmedia-bot/src/config.rs` | Manual env parsing into `Config` |
|
||||
| `crates/xmedia-bot/src/handlers.rs` | `Command` enum (teloxide `BotCommands`), message/inline/callback handlers, URL extraction, global statics; per-URL work spawned with a `Semaphore(8)` cap (teloxide's per-chat workers are sequential — batch-forwards need concurrency) |
|
||||
| `crates/xmedia-bot/src/db.rs` | `DbPool`: one shared SQLite connection pool (`POOL_SIZE = 4`, WAL, busy_timeout) for all three tables over `$DATA_DIR/task_queue.db` (default `data/`) — the three stores share it; `open_store` creates file + schema, `with_conn` runs all rusqlite I/O in `spawn_blocking` |
|
||||
| `crates/xmedia-bot/src/handlers/` | Handler modules: `mod.rs` (message entry point, `reply`, `log_key`), `commands.rs` (teloxide `BotCommands` enum + command executor, incl. `/test <url>` (send-only) / `/debug <url>` (parse-only) and the admin-only `/bot_dict` state dump), `urls.rs` (URL extraction + bounded job channel (256) drained by `URL_WORKERS = 8` workers (`start_url_workers`) — backpressure instead of unbounded spawns; teloxide's per-chat workers are sequential — batch-forwards need concurrency), `inline.rs`/`callback.rs` (inline queries / edit-before-forward buttons), `statics.rs` (global statics) |
|
||||
| `crates/xmedia-bot/src/state.rs` | `ChatStore`: parking_lot `Mutex<HashMap>` cache + SQLite write-through (`chat_state` table) |
|
||||
| `crates/xmedia-bot/src/link_cache.rs` | `LinkCache`: SQLite-backed cache (`link_cache` table) of successfully sent posts — raw caption fields + Telegram `file_id`s; repeat links re-send locally (no fetch/upload), TTL + prune, invalidated on permanent send failure |
|
||||
| `crates/xmedia-bot/src/queue.rs` | `PersistentTaskQueue`: SQLite-backed queue (`tasks` table), `QUEUE_WORKERS = 4` concurrent workers (lease via `BEGIN IMMEDIATE` + `locked_until` TTL), retry→dead-letter, `Notify::notify_waiters` wakeup, `busy_timeout` on all connections |
|
||||
| `crates/xmedia-bot/src/send.rs` | Media senders, upload fallback, error classification, queue task handlers |
|
||||
| `crates/xmedia-bot/src/queue.rs` | `PersistentTaskQueue`: SQLite-backed queue (`tasks` table), `QUEUE_WORKERS = 4` concurrent workers (lease via `BEGIN IMMEDIATE` + `locked_until` TTL), retry→dead-letter, `notify_one` worker wakeup plus a separate `Notify` for the 30 s lease-expiry sweep (a shared one let the sweep steal the workers' wakeup permit), `busy_timeout` on all connections |
|
||||
| `crates/xmedia-bot/src/ctx.rs` | `AppContext`: the injected collaborators (`sender` + `ChatStore`/`PersistentTaskQueue`/`LinkCache`/`Config`), `from_statics` for production and the `CONTEXT` static the worker closures hold. `test_support::TestStores` backs handler tests with a tempdir store set |
|
||||
| `crates/xmedia-bot/src/send/` | `send/mod.rs`: `Task`/`MediaItemPayload` payloads, `SendError`/`Classification`, `send_media_sequence`/`send_animation`/`forward_messages`; `send/input_media.rs`: payload → `InputFile`/`InputMedia` + `build_media_group` (caption on the first item only); `send/upload.rs`: the download-and-reupload fallback (`prepare_upload_item`/`send_batch_via_upload`, photo downscale handoff); `send/post_send.rs`: link-cache write, `KEEP_ALIVE` registry, `settle_task`, `post_send_actions`, `handle_task`/`dead_letter_notify` |
|
||||
| `crates/xmedia-bot/src/media_sender.rs` | `MediaSender` trait: the user-flow surface (`send_media_group`/`send_animation`/`copy_messages`/`send_message`/`answer_callback_query`/`edit_message_caption`/`delete_message`/`send_chat_action`) implemented by teloxide `Bot` (per-chat rate-limited) and by a recording `MockSender` in tests. Admin/setup APIs (`get_chat`, `set_my_commands`, …) stay on the concrete `Bot` |
|
||||
| `crates/xmedia-bot/src/rate_limit.rs` | Per-chat token bucket (`CAPACITY = 20`, ~20 msg/min refill) paced before sends reach the API so batch forwards don't trip flood control |
|
||||
|
||||
## Development Commands
|
||||
|
||||
@@ -48,52 +56,53 @@ cargo clippy --workspace --all-targets # lint (Clippy is the configured IDE lint
|
||||
cargo fmt --check # formatting
|
||||
```
|
||||
|
||||
Docker: `docker build -t tgxmb .` then `docker run --rm -d --name tgxmb --env-file .env -v ./data:/app/data tgxmb`. Runtime requires **ffmpeg** (built into the image).
|
||||
Docker: `docker build -t tgxmb .` then `docker run --rm -d --name tgxmb --env-file .env -v ./data:/app/data tgxmb`. Runtime requires **ffmpeg** (built into the image). The builder fetches crates.io + ffmpeg; on restricted networks pass proxy build args, e.g. `--build-arg HTTP_PROXY=http://host.docker.internal:10808 --build-arg HTTPS_PROXY=…` (Docker Desktop builds can't reach the host loopback — use `host.docker.internal`).
|
||||
|
||||
## Code Conventions & Common Patterns
|
||||
|
||||
- **No anyhow/thiserror.** Errors are hand-rolled enums with manual `Display`/`source()`/`From` impls: `QueueError` (`Retryable { delay_seconds, payload }` / `Permanent`), `SendError` (Retryable/Permanent), `FetchError` (`Http`/`Json`/`Pixiv`/`NotFound`/`Blocked`), `PixivError`, `Classification`. New errors should follow this pattern.
|
||||
- **Global state via `std::sync::LazyLock` statics**, not DI: `CONFIG`, `CHAT_STORE`, `TASK_QUEUE` in `handlers.rs`; shared reqwest `CLIENT` in `x-media/src/site/mod.rs`. `Bot` is passed/cloned into handlers; queue workers rebuild `Bot::from_env()`.
|
||||
- **Errors via `thiserror` derive** (no anyhow): the public, stringified errors — `FetchError` (`Http`/`Json`/`Pixiv`/`Site`/`NotFound`/`Blocked`) and `PixivError` — derive `thiserror::Error` with `#[from]` conversions; `Display`/`source()` come from the derive. The internal control-flow enums — `QueueError` (`Retryable { delay_seconds, payload }` / `Permanent`), `SendError` (Retryable/Permanent), `Classification`, `FallbackError` — carry no `Display` and are handled by direct variant matching. New errors should follow the same split: stringified/public errors derive `thiserror`, internal flow enums stay plain.
|
||||
- **Global state via `std::sync::LazyLock` statics**, not DI: `CONFIG`, `CHAT_STORE`, `TASK_QUEUE` in `handlers/statics.rs`; shared reqwest `CLIENT` in `x-media/src/site/mod.rs`. `Bot` is passed/cloned into handlers; queue workers share the process-wide `send::BOT` (`LazyLock<Bot>`, force-initialized in `main` so a missing token fails at startup).
|
||||
- **Async**: tokio multi-thread runtime (`#[tokio::main]` default). All rusqlite I/O inside `tokio::task::spawn_blocking`. Long loops use `tokio::select!` with `tokio::sync::{watch, Notify}` stop/wake channels. No streams.
|
||||
- **Blocking sync primitives**: `parking_lot::Mutex` for hot caches, `tokio::sync::Mutex` for async-shared state (pixiv token cache), `AtomicBool` for feature gates.
|
||||
- **Site adapter convention** (no trait, no enum dispatch — follow the existing convention): each site module exports `PATTERN: LazyLock<Regex>`, `enabled() -> bool`, `fetch_from_url(url) -> Result<Fetched, FetchError>`; `site/mod.rs` re-exports the site struct and `fetch_once` adds one guarded if-branch. Adding a site = new `site/<name>/{mod.rs,interface.rs,model.rs}` + one branch in `fetch_once`.
|
||||
- **Site adapter convention**: each site module exports `PATTERN: LazyLock<Regex>`, `enabled() -> bool`, `fetch_from_url(url) -> Result<Fetched, FetchError>`, plus `cache_key`/`is_retryable`/`media_headers`, and a unit struct `<Name>Site` implementing `site::Site`; the central dispatcher (`site/mod.rs`) only iterates the `SITES` registry. Adding a site = new `site/<name>/{mod.rs,interface.rs,model.rs}` + one `Box::new(...)` entry in `SITES` — the bot crate never lists sites (SetFormat whitelist, cache-key site lookup and startup validation all derive from the registry). Async trait methods return `SiteFuture` (a boxed `Pin<Box<dyn Future + Send>>`) because `async fn` in traits is not dyn-compatible.
|
||||
- **Serde**: per-site `model.rs` are pure `Deserialize` DTOs mirroring API JSON; site structs in `interface.rs` have private fields, a `caption()` builder, and `impl From<SiteStruct> for Fetched`. Persisted payloads use internally-tagged enums (`#[serde(tag = "kind")]` / `type`).
|
||||
- **Naming**: module-per-concern, snake_case files, `CamelCase` types, `snake_case` fns. `//!` module docs and `///` docs on non-obvious logic (syndication token, ugoira encoding, `display_text_range`).
|
||||
- **Retries**: only `x-media::site::fetch` retries (3 attempts, `1 << attempt` backoff, HTTP errors only). Queue retries are explicit `QueueError::Retryable` with computed delay (`retry_delay_seconds`).
|
||||
- Logging via `log` macros (`pretty_env_logger`, level from `RUST_LOG`).
|
||||
- **Retries**: only `x-media::site::fetch` retries (3 attempts, `1 << attempt` backoff, HTTP errors only); `site::fetch_once` is the same code path with a single attempt, used by inline queries whose answer window is shorter than the backoff. Queue retries are explicit `QueueError::Retryable` with computed delay (`retry_delay_seconds`).
|
||||
- Logging via `log` macros (`pretty_env_logger`, level from `RUST_LOG`). Level convention: `info` = lifecycle + per-post business results (`sent`/`forwarded`/`copied`), admin/operator actions and anomalies (fallback, retry enqueue, dead-letter is `error`); `debug` = per-request detail (message/command/URL extraction, `fetching`/`fetched`, batch sends, queue processing, photo processing, inline queries). Full user-submitted URLs and message text only appear at `debug`; at `info` and above links are printed via the normalized cache key (`handlers::log_key`, e.g. `[key=twitter:123...]`) so logs stay short and do not echo user data.
|
||||
|
||||
## Important Files
|
||||
|
||||
| File | Why it matters |
|
||||
|---|---|
|
||||
| `crates/xmedia-bot/src/main.rs` | Startup sequence, webhook vs polling, graceful shutdown (SIGINT via teloxide ctrlc / SIGTERM via `stop_token` for docker, → sweep stop → admin msg → queue stop) |
|
||||
| `crates/xmedia-bot/src/handlers.rs` | `CHAT_STORE`/`TASK_QUEUE`/`CONFIG` singletons (open `data/task_queue.db` **relative to CWD**); command dispatch; URL extraction; retry enqueue |
|
||||
| `crates/xmedia-bot/src/send.rs` | Constants `MAX_MEDIA_GROUP = 9`; fallback chain; `classify_request_error`; download-and-reupload fallback triggered only by Telegram API errors (`is_media_fetch_failure` / `is_size_error`) |
|
||||
| `crates/xmedia-bot/src/handlers/` | `statics.rs` = `CHAT_STORE`/`TASK_QUEUE`/`CONFIG` singletons (open `$DATA_DIR/task_queue.db`, default `data/` **relative to CWD**, dir auto-created); `commands.rs` = command dispatch (incl. `/test <url>` send-only, `/debug <url>` parse-only, and the admin-only `/bot_dict` state dump); `urls.rs` = URL extraction + the per-URL pipeline (`url_media` takes a `PostSend` mode: chat settings vs `/test`'s suppressed actions); `inline.rs` = debounced inline queries; `callback.rs` = edit-before-forward buttons (dptree entry + testable `handle_callback` core) |
|
||||
| `crates/xmedia-bot/src/send/` | `mod.rs`: constants `MAX_MEDIA_GROUP = 9`; `classify_request_error`; the senders. `upload.rs`: download-and-reupload fallback triggered only by Telegram API errors (`is_media_fetch_failure` / `is_size_error`). `post_send.rs`: settlement (`settle_task`), cache write, post-send actions, queue handlers. `input_media.rs`: payload → `InputMedia` |
|
||||
| `crates/xmedia-bot/src/photo.rs` | Pure-Rust photo processing (no ffmpeg): `png` (image-png) decode/encode + `zune-jpeg` decode + `fast_image_resize` Lanczos3 downscale + `jpeg-encoder`. Photos over Telegram's limits (width + height > 10000 px → `PHOTO_INVALID_DIMENSIONS`; bytes > 10 MiB) are decoded, downscaled keeping the format, PNG bit depth > 24 (RGBA 32-bit / 16-bit per channel) reduced to 24-bit RGB with alpha flattened white (≤24-bit untouched, never upconverted), and transcoded to JPEG only if still over the cap; memory budget guarded, otherwise the item's smaller fallback URL |
|
||||
| `crates/x-media/src/site/mod.rs` | Dispatcher, `Fetched`/`FetchError`, shared `CLIENT`, `download_media` (adds `Referer: https://www.pixiv.net/` for `pximg.net` hotlink protection) |
|
||||
| `crates/x-media/src/site/pixiv/api.rs` | OAuth token exchange (hardcoded app client id/secret), access-token cache, ugoira zip→MP4 via ffmpeg in `spawn_blocking` |
|
||||
| `Dockerfile` | Multi-stage: cached dep layer via stub sources + `touch *.rs` mtime hack, static ffmpeg from ffmpeg.martin-riedl.de (`FFMPEG_URL` arg, `unzip -t` integrity check), `debian:bookworm-slim` runtime, entrypoint |
|
||||
| `Dockerfile` | Multi-stage: cached dep layer via stub sources + `touch *.rs` mtime bump (cargo's freshness is mtime-based and `cargo clean -p` removes 0 files — the touch is what forces the real sources to rebuild while deps stay cached), static ffmpeg from ffmpeg.martin-riedl.de (`FFMPEG_URL` arg, optional `FFMPEG_SHA256` checksum, `unzip -t` integrity check), `debian:bookworm-slim` runtime, entrypoint. Runtime ships **no libssl/libcrypto/CA bundle** — rustls webpki-roots handles all TLS, and the static ffmpeg only processes local files (downloads go through reqwest) |
|
||||
| `docker-entrypoint.sh` | Privilege drop: `useradd` with `LOCAL_USER_ID` (default 9001) + `setpriv` (no gosu on bookworm-slim) |
|
||||
| `docker-compose.yml.example` | Deployment env reference (real `docker-compose.yml` is gitignored). Ships nginx-proxy + acme-companion: webhook mode needs TLS termination in front (teloxide's axum listener is HTTP-only; `WEBHOOK_CERT` only feeds `set_webhook`), bot exposes `VIRTUAL_HOST`/`VIRTUAL_PORT` on the shared `proxy` network, no host port; container names `nginx-proxy`/`acme-companion`/`tgxmb`, start order via `depends_on` (proxy → acme → bot) |
|
||||
| `.github/workflows/docker.yml` | CI: build+push to Docker Hub on tag `v*`/master; **no test step**; buildx gha cache (`cache-from`/`cache-to`, scope `tgxmb-build`, `mode=max`) so cargo deps + ffmpeg layers are restored across runs |
|
||||
| `.github/workflows/docker.yml` | CI: build+push to Docker Hub on tag `v*`/master, plus a build-only check on PRs touching the build inputs; **no test step**; verifies a release tag matches both crate versions; buildx gha cache (`cache-from` always, `cache-to` except on PRs, scope `tgxmb-build`, `mode=max`) so cargo deps + ffmpeg layers are restored across runs; `FFMPEG_URL`/`FFMPEG_SHA256` come from repo variables when set |
|
||||
| `README.md` | Feature docs + command table (Chinese) |
|
||||
|
||||
## Runtime/Tooling Preferences
|
||||
|
||||
- **Rust, stable, edition 2024**, workspace resolver 3. No `rust-version`/MSRV pin, no `rust-toolchain.toml` — recent stable is assumed. No nightly features.
|
||||
- Package manager: **Cargo** (workspace with path dep `x-media` ← `xmedia-bot`). No `[workspace.package]`/shared deps — each crate lists deps independently.
|
||||
- **Two reqwest versions coexist in the lock** (0.12.28 via teloxide, 0.13.3 in x-media) — don't unify casually.
|
||||
- Config is **environment-variable driven** (dotenv loads `.env`, gitignored; no `.env.example` exists). Key vars: `TELOXIDE_TOKEN` (required), `PIXIV_REFRESH_TOKEN`, `TWITTER_AUTH_TOKEN` (optional; x.com `auth_token` cookie — enables the logged-in GraphQL fallback that fetches NSFW tweets syndication withholds), `BOT_ADMIN` (comma-separated ids), `EDIT_MESSAGE_TTL_SECONDS` (default 86400), `LINK_CACHE_TTL_SECONDS` (default 604800), `WEBHOOK`/`WEBHOOK_URL`/`WEBHOOK_LISTEN`/`WEBHOOK_PORT`/`WEBHOOK_CERT`/`WEBHOOK_SECRET_TOKEN` (webhook mode requires URL/listen/port, `.expect`ed; `WEBHOOK_CERT` is Telegram-facing self-signed validation only — TLS must be terminated by a reverse proxy), `RUST_LOG`, `TELOXIDE_PROXY`, `LOCAL_USER_ID` (entrypoint only).
|
||||
- SQLite via `rusqlite` with `bundled` feature (no system libsqlite needed). DB file `data/task_queue.db` is CWD-relative — run from the workspace root, or `/app` in Docker. Mount `./data` and `./cert` volumes.
|
||||
- **TLS is rustls end-to-end** (no native-tls/openssl in the tree, no libssl in the Docker runtime image): `teloxide` is declared `default-features = false` with `["webhooks-axum", "macros", "rustls", "ctrlc_handler"]` (the removed `default` also carried `native-tls` and `ctrlc_handler` — the latter must stay); x-media's reqwest is `default-features = false` with `["json", "rustls-tls"]` (webpki-roots baked in, so the image ships no CA bundle). One reqwest 0.12.28 in the lock.
|
||||
- **Versioning**: bump the version in all three places (`crates/x-media/Cargo.toml`, `crates/xmedia-bot/Cargo.toml`, `Cargo.lock`) and **keep `README.md`, `README.en.md` and `AGENTS.md` in sync with the code on every bump**, then commit (`chore: bump version to X.Y.Z`), create an annotated tag `vX.Y.Z`, and push branch + tag (the tag push triggers the Docker Hub build). The tag must equal both crate versions: `.github/workflows/docker.yml` verifies that before building, and `--locked` verifies the lock file.
|
||||
- Config is **environment-variable driven** (dotenv loads `.env`, gitignored; no `.env.example` exists). Key vars: `TELOXIDE_TOKEN` (required), `PIXIV_REFRESH_TOKEN`, `TWITTER_AUTH_TOKEN` (optional; x.com `auth_token` cookie — enables the logged-in GraphQL fallback that fetches NSFW tweets syndication withholds), `BILIBILI_COOKIE` (optional; whole bilibili cookie string — bilibili dynamics fetch anonymously and add their own device cookies, this only rescues an egress IP that bilibili has hard-flagged with `-352`/412), `BOT_ADMIN` (comma-separated ids), `EDIT_MESSAGE_TTL_SECONDS` (default 86400), `LINK_CACHE_TTL_SECONDS` (default 604800), `CAPTION_QUOTE_TEXT_CHARS` (default 200; a post whose text — the `title` plus `content` joined, see `site::compose_text` — reaches this length gets that text wrapped in an expandable blockquote inside its caption, the URL and author line staying outside; `0` disables it. Applied at the send boundary in `send::quote_long_caption`, which locates the text as what follows the author link, so a `/set_format` that moves `{title}`/`{content}` elsewhere and pixiv's title-inside-a-link layout opt out; `copy_messages` forwards and queued retries inherit the wrap, while the edit-before-forward rewrite stays unquoted by design), `DATA_DIR` (default `data`, CWD-relative; the SQLite dir, auto-created), `WEBHOOK`/`WEBHOOK_URL`/`WEBHOOK_LISTEN`/`WEBHOOK_PORT`/`WEBHOOK_CERT`/`WEBHOOK_SECRET_TOKEN` (webhook mode requires URL/listen/port, `.expect`ed; `WEBHOOK_CERT` is Telegram-facing self-signed validation only — TLS must be terminated by a reverse proxy), `RUST_LOG`, `TELOXIDE_PROXY`, `LOCAL_USER_ID` (entrypoint only).
|
||||
- SQLite via `rusqlite` with `bundled` feature (no system libsqlite needed). DB file `$DATA_DIR/task_queue.db` (default `data/task_queue.db`, CWD-relative — run from the workspace root, or `/app` in Docker; set `DATA_DIR` to pin state anywhere). Mount `./data` and `./cert` volumes.
|
||||
- `.gitattributes` enforces LF for `*.sh` (CRLF breaks shebangs in containers). `.gitignore`: `.env`, `data/`, `cert/`, `docker-compose.yml`, `/target`, `.idea/`.
|
||||
- Docs are in Chinese; user-facing bot strings too. Keep that convention when editing captions/templates/docs.
|
||||
- Docs are in Chinese (README, AGENTS.md); user-facing bot strings are in English. Keep that split when editing user-facing strings and docs.
|
||||
|
||||
## Testing & QA
|
||||
|
||||
- **~51 tests, all inline `#[cfg(test)] mod tests`** — no `tests/` integration directories. Framework: built-in Rust test + `#[tokio::test]` (dev-deps only in `x-media`: tokio macros/rt-multi-thread, dotenv).
|
||||
- **~180 tests, all inline `#[cfg(test)] mod tests`** — no `tests/` integration directories. Framework: built-in Rust test + `#[tokio::test]` (dev-deps only in `x-media`: tokio macros/rt-multi-thread, dotenv).
|
||||
- No mocking framework anywhere (no mockito/wiremock/mockall). Conventions: pure-function units (regex parsing, serde round-trips, chunking, retry math) tested synchronously; async tests use real dependencies — file-backed SQLite via `tempfile` (`queue.rs::new_queue()` helper), live network fetches.
|
||||
- Live-network tests exist in `site/twitter/interface.rs` (3), `site/bsky/interface.rs` (2), `site/pixiv/interface.rs`/`api.rs` (env-gated on `PIXIV_REFRESH_TOKEN`/dotenv, skip by early return). Run the full suite with `cargo test --workspace`.
|
||||
- Live-network tests exist in `site/twitter/interface.rs` (5), `site/bsky/interface.rs` (2), `site/misskey/interface.rs` (1), `site/bilibili/interface.rs` (4), `site/pixiv/api.rs` (1); `photo.rs` adds one `#[ignore = "heavy: …"]` test. `site/mod.rs` also has a **token-gated but not `#[ignore]`d** pixiv download test (`download_media_pixiv_original_with_referer`): it hits `i.pximg.net` whenever `PIXIV_REFRESH_TOKEN` is set, so a local `cargo test --workspace` is not fully offline and can flake on a pixiv CDN body timeout. Test gating convention (enforced by `.github/workflows/ci.yml`): pure unit tests always run; live-network tests carry `#[ignore = "live network: ..."]` (run via `cargo test --workspace -- --ignored live`); token-gated pixiv tests early-return when `PIXIV_REFRESH_TOKEN` is absent **or empty** (an unset GitHub secret arrives as `""` — `is_err()` alone would run them tokenless and fail), and the bilibili live tests early-return when the API answers risk control (`-352`, which bilibili applies per IP by request volume). Run the full offline suite with `cargo test --workspace`.
|
||||
- Fixtures are inline `serde_json::json!` builder fns (`fixture()`, `thread_json()`, `illust_json()`), not files. The shared `CLIENT` sets `pool_max_idle_per_host(0)` under `#[cfg(test)]` to avoid cross-runtime `DispatchGone`.
|
||||
- **CI runs no tests** — `.github/workflows/docker.yml` only builds/pushes the image; verification is a local responsibility.
|
||||
- Untested and hard to test without a mock seam: `handlers.rs` (depends directly on teloxide `Bot`); `main.rs`, `config.rs`, `state.rs`; `media.rs`, `lib.rs`, all `model.rs`.
|
||||
- No coverage tracking, no lint gate in CI.
|
||||
- **CI** — `.github/workflows/ci.yml` (actions pinned to commit SHAs, `--locked` on every cargo invocation, `concurrency` cancels superseded runs, `RUST_BACKTRACE=1`) runs `cargo fmt --check` + `cargo clippy --workspace --all-targets --locked -- -D warnings` + `cargo test --workspace --locked` + a release-profile `cargo build --release --locked` + an `actions-rust-lang/audit` dependency-vulnerability gate (offline, no secrets, on every push/PR) and a `live` job (schedule/manual/tag only, `-p x-media` since every network/secret-gated test lives there, `continue-on-error`) for the `#[ignore]`d live + token tests. `.github/workflows/docker.yml` builds and pushes the image on master/tag and runs a **build-only check on pull requests touching the build inputs** (`Dockerfile`, entrypoint, manifests, `.dockerignore`); a release tag must match both crate versions or the build stops, and `FFMPEG_URL`/`FFMPEG_SHA256` are taken from repository variables when set (a release can pin an exact ffmpeg build). `.github/dependabot.yml` keeps crates, the pinned actions and the Docker base images current.
|
||||
- Untested and hard to test without a mock seam: `main.rs`, `config.rs`, `db.rs`, `handlers/statics.rs`, `media_sender.rs` (holds the `MockSender` itself); in `x-media`: `media.rs`, `lib.rs`, all `model.rs`. The `commands.rs` *executor* needs a real `Bot` (only its pure report builder is tested). Everything else — `handlers/{mod,callback,inline,urls}.rs`, `send/*`, `ctx.rs`, `state.rs`, `queue.rs`, `link_cache.rs`, `rate_limit.rs` — is driven through `TestStores`/`ctx::test_support` and the scripted `MockSender`.
|
||||
- No coverage tracking.
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
# Bilibili 动态支持:研究与实现记录
|
||||
|
||||
状态:已实现(`crates/x-media/src/site/bilibili/`)。本文记录上游调研、实测数据与最终设计;
|
||||
长期契约以 `AGENTS.md` 为准。
|
||||
|
||||
范围:**只发动态里的图片与动图**。动态内嵌视频不发流,降级为封面图;`b23.tv` 短链不匹配;
|
||||
视频页 / 番剧 / 直播间 / 专栏 / 音频均不支持。
|
||||
|
||||
---
|
||||
|
||||
## 1. 上游实现研究
|
||||
|
||||
### 1.1 nazurin(`nazurin/sites/bilibili/`,4 个文件 ~6 KB)
|
||||
|
||||
- 入口正则:`t\.bilibili\.com/(\d+)`、`t\.bilibili\.com/h5/dynamic/detail/(\d+)`、`bilibili\.com/opus/(\d+)`。
|
||||
- 请求:`GET https://api.bilibili.com/x/polymer/web-dynamic/v1/detail?id={id}`,仅加 `Referer: https://t.bilibili.com/{id}`。
|
||||
**无 cookie、无 WBI 签名、无 `build` 参数**。
|
||||
- 错误:`code == 4101147` → not found;`code != 0` 或缺 `data` → 报错。
|
||||
- 媒体:只取 `item.modules.module_dynamic.major.draw.items[].src`;缩略图 `src + "@518w.jpg"`;
|
||||
`size` 字段单位是 **KB**。`major` 为空或 `draw.items` 为空 → "No image found"。
|
||||
**忽略视频、转发(forward)与纯文字动态**。
|
||||
- caption:`"#" + module_author.name` + `module_dynamic.desc.text`,链接写死 `https://www.bilibili.com/opus/{id}`。
|
||||
|
||||
### 1.2 telegram-bili-feed-helper(`biliparser/provider/bilibili/`,9 个文件 ~57 KB)
|
||||
|
||||
- 9 个策略类(Video/Opus/Live/Audio/Read + Feed 基类 + Credential + api 工具):门禁正则
|
||||
`bilibili\.com|b23\.tv|BV\w{10}|av\d+`,再分流,兜底 `client.head(url)` 跟随重定向后按子串分流。
|
||||
- 动态:`GET /x/polymer/web-dynamic/desktop/v1/detail?id={id}&build=11605`(**单条,无分页**);
|
||||
客户端带桌面 UA、随机 `buvid3={uuid}infoc`;登录态用 `bilibili-api-python` 的 `Credential`
|
||||
(Redis 持久化 `SESSDATA/bili_jct/buvid3/buvid4/ac_time_value/DedeUserID`,扫码登录)。
|
||||
- **同样没有 WBI 签名 / appkey 签名**:playurl 用的是非 WBI 的 `/x/player/playurl`。
|
||||
- 媒体:`major.type` 分派 —— DRAW 取全部 `items[].src`;ARCHIVE/PGC/ARTICLE/MUSIC/COMMON/LIVE
|
||||
只取一张 `cover`;FORWARD 取原动态作者/正文并递归进 `orig` 找媒体。
|
||||
- 视频:仅独立 video 策略解析(`qn` 720P→480P→360P 试 durl,再退 DASH + ffmpeg 合并);
|
||||
**动态内嵌视频只发封面**。
|
||||
- 错误:要求 `status==200 && code==0`;风控 `-352`/`-412` 无特殊处理。
|
||||
|
||||
### 1.3 取舍
|
||||
|
||||
| 维度 | nazurin | bff | 本仓库 |
|
||||
|---|---|---|---|
|
||||
| 接口 | `v1/detail?id=` | `desktop/v1/detail?id=&build=` | `v1/detail?id=`(实测可用) |
|
||||
| 认证 | 无 | buvid3 + SESSDATA | 默认匿名;可选 `BILIBILI_COOKIE` |
|
||||
| WBI | 无 | 无 | 不实现(无需求) |
|
||||
| 图片 | `major.draw.items` | 同 + forward 递归 | 同,加 `orig` 递归、`http→https`、`.gif → Animated` |
|
||||
| 视频 | 完全忽略 | 动态内嵌视频发封面 | 发封面(不发流) |
|
||||
| 短链 | 不匹配 | 跟随重定向 | 不匹配(多数短链是视频,会让"静默忽略"变成失败提示) |
|
||||
|
||||
---
|
||||
|
||||
## 2. 实测验证(2026-09-17,真实请求)
|
||||
|
||||
| 验证项 | 结果 |
|
||||
|---|---|
|
||||
| `v1/detail?id=`(无 cookie、UA `Mozilla/5.0`、带 Referer) | `200 {"code":0}` ✅ |
|
||||
| 同上,不带 cookie 也不带 Referer | `200 {"code":0}` ✅(无强制鉴权) |
|
||||
| bff 的 `bilibili_pc/…Electron/22.3.27` UA | `code:-352` ❌ → **不要抄它的 UA** |
|
||||
| `desktop/v1/detail?build=11605` | `code:-352` ❌ |
|
||||
| `feed/space?host_mid=`(用户时间线) | 首次成功、随后 `-352`,也见过 HTTP 412 → **不碰** |
|
||||
| 不存在 / 已删除的动态 | `code:500` "Cannot read property 'only_fans' of undefined"(nazurin 的 4101147 已失效) |
|
||||
| 非数字 id | `code:-400` param parsing failed |
|
||||
| 图片 `i0.hdslb.com/bfs/new_dyn/*.jpg` | `HEAD 200 image/jpeg`,带/不带 Referer 均可;`+@518w.jpg` → 25–42 KB ✅ |
|
||||
| `t.bilibili.com/h5/dynamic/detail/<id>` | `200` ✅ |
|
||||
| `m.bilibili.com/dynamic/<id>` | `302 → t.bilibili.com/<id>` ✅ |
|
||||
| `www.bilibili.com/opus/<id>` | `200`,转发动态 `302 → t.bilibili.com/<id>` ✅ |
|
||||
| `b23.tv/BV1JTtt6JEZu` | `302 → www.bilibili.com/video/BV…`(视频) |
|
||||
| `b23.tv/<无效码>` | **HTTP 200** + `{"code":-404}` ⚠️ 短链判定不能只看状态码 |
|
||||
| `playurl`(仅调研用,未采用) | `fnval=1` 匿名给 durl:720P=9.18 MiB / 360P=2.97 MiB;`fnval=4048` 匿名 DASH 上限仅 480P |
|
||||
| `dyn_archive` 字段 | 有 `aid/bvid/cover/title/duration_text`,**没有 `cid`**(所以发流要再来一次 `view` 请求) |
|
||||
| **风控阶梯(同一 IP 连续请求后实测)** | ① 无 cookie → `-352`;② 仅 `buvid3` → 仍 `-352`;③ `buvid3`+`buvid4`(取自匿名 `/x/frontend/finger/spi`)→ **`code:0` 恢复**;④ 继续高频请求后 → 连同 buvid 一起 `-352`(此时只有登录 cookie 或换 IP) |
|
||||
| **正文位置(24 条真实动态逐条审计)** | 有正文的动态都在 `module_dynamic.desc.text`(图文/转发/纯文字,含 34–193 字样本);**AV(视频投稿)动态 `desc` 恒为 `null`**,内容在 `major.archive.title` / `.desc` 卡片里 → 已做 title 回退 |
|
||||
| **`features=itemOpusStyle` 的效果** | 同一端点带此参数后,图文帖改为 `major.opus` 形态:`pics[]`(图,key 是 `url`)、`summary.text`(正文,未截断,实测 307 字整段)、`title`(可选标题);不带参数则是 legacy `major.draw` + `desc`,而 **opus 图文帖的 `desc` 为 `null`、正文与标题完全丢失**(`opus/1248857553488576532`:legacy `desc:null`,带参数 `summary.text="[doge_金箍]黑白搭配"`)。AV / 转发帖不受该参数影响 → 适配器改为请求时带参数,并保留 legacy 形态兜底 |
|
||||
| feed 与 detail 的差异 | `feed/space` 的 item 会把 `desc.text` 挖空,**只有 detail 有正文** → 排查时不要用 feed 数据判断正文缺失 |
|
||||
| 不存在的 19 位 id | `4101105 请求数据发生错误`(提示可重试,但只出现在不可能存在的 id 上)→ 仍归入永久错误,见 `code_error` 注释 |
|
||||
|
||||
测试样本(live 测试用):
|
||||
|
||||
| 样本 | id | 期望 |
|
||||
|---|---|---|
|
||||
| 图片动态(2 图 + 话题) | `1245284537985925159` | 2 个 `Illustration`,`{tags}` = `ALin出道20周年快乐` |
|
||||
| 转发动态 | `1248982077447077907` | 媒体来自 `orig`(1 图),正文可含 `//@` |
|
||||
| 视频动态 | `1248717597691609105` | 封面 1 张 `Illustration` |
|
||||
| 纯文字动态 | `1246767523595026450` | `media` 为空 |
|
||||
|
||||
关键字段路径:
|
||||
|
||||
```
|
||||
data.item.id_str
|
||||
data.item.modules.module_author.{name,mid}
|
||||
data.item.modules.module_dynamic.desc.text
|
||||
data.item.modules.module_dynamic.topic.{id,name} # 单话题,{tags} 来源
|
||||
data.item.modules.module_dynamic.major.{draw.items[].src, archive.cover}
|
||||
data.item.orig # 转发时存在,结构与 item 相同
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. 实现
|
||||
|
||||
```
|
||||
crates/x-media/src/site/bilibili/mod.rs # re-export
|
||||
crates/x-media/src/site/bilibili/interface.rs # PATTERN / cache_key / enabled / is_retryable /
|
||||
# media_headers / BilibiliSite / fetch / code_error /
|
||||
# From<Item> for Fetched / caption / 12 单测 + 2 live
|
||||
crates/x-media/src/site/bilibili/model.rs # 纯 Deserialize DTO(全 Option)
|
||||
```
|
||||
|
||||
- **正则**(同时用于分发、抽 id、缓存键,一个正则三用):
|
||||
`^(?:https?://)?(?:www|t|m)\.bilibili\.com/(?:opus/|dynamic/|h5/dynamic/detail/)?(\d+)`
|
||||
- **缓存键**:`bilibili:<动态 id>`;`source_url` 统一 `https://www.bilibili.com/opus/{id}`。
|
||||
- **请求**:`GET /x/polymer/web-dynamic/v1/detail?id=` + `Referer: https://www.bilibili.com/`;
|
||||
`Cookie` 头按优先级取:`BILIBILI_COOKIE` → 缓存的设备 cookie(`GET /x/frontend/finger/spi` 取 `buvid3`/`buvid4`,
|
||||
进程内缓存一次;取不到就不带 cookie,仅 debug 日志)→ 无。指纹接口本身失败**不**让抓取失败。
|
||||
走共享 `CLIENT`(UA `Mozilla/5.0`,30s 超时,`TELOXIDE_PROXY` 透传)。
|
||||
- **错误映射**:`0` → 成功;`-352/-412` 与 HTTP 412 → `Transient`(可重试,队列退避;首次记一条 warn 提示
|
||||
`BILIBILI_COOKIE`);`500`/`4101147` → `NotFound`(永久);其他 code → `Site`(永久)。
|
||||
- **媒体**:
|
||||
- `major.opus.pics[]`(带 `features=itemOpusStyle` 时的图文帖形态,字段名是 `url`)→ 每张一张图;
|
||||
其次 `major.draw.items[]`(legacy,字段名 `src`)→ 同样逐张;`http://` / `//` → `https://`,非 https 开头直接丢弃。
|
||||
`.gif` → `Media::Animated`(`thumbnail_url` 留空,Telegram 自己取首帧——`@518w.jpg` 只对 jpg/webp 实测过),
|
||||
其余 → `Media::Illustration`(`thumbnail_url = url + "@518w.jpg"`,兼作超大时的降级 URL)。
|
||||
- `major.archive.cover` → 1 张 `Illustration`(视频不发流)。
|
||||
- 转发且自身无媒体 → 递归取 `orig` 的媒体;正文拼 `//@{原作者}:\n{原文}`。
|
||||
- 其他 major(PGC/ARTICLE/MUSIC/LIVE/COMMON)不建模 → 无媒体,走既有 "No media found"。
|
||||
- **正文 / title**(按信息量从多到少回退):`major.opus.title` + `major.opus.summary.text`
|
||||
→ `module_dynamic.desc.text` → `major.archive.title`。三者分别对应:图文文档(标题+正文)、
|
||||
legacy/转发帖正文、视频投稿卡片标题。开头结尾空白做 trim;整体再由既有 `truncate_caption` 截断。
|
||||
- **caption**(与 misskey 同形):`{opus 链接}\n<a href="space.bilibili.com/{mid}">{name}</a>: {正文}`;
|
||||
`RenderData` 的 `{tags}` 来自话题名;正文由既有 `truncate_caption` 截断。
|
||||
- **注册表**:`SITES` 末尾追加 → `/set_format` 白名单、链接缓存、启动校验、日志前缀全部自动生效。
|
||||
- **bot 侧仅文案**:`handlers/commands.rs` 三处站点清单字符串 + `state.rs`/`handlers/mod.rs` 注释。
|
||||
|
||||
### 与原计划的偏差(及原因)
|
||||
|
||||
| 原计划 | 实际 | 原因 |
|
||||
|---|---|---|
|
||||
| `x/web-interface/view` + `playurl` 发视频 | 不做 | 需求收窄为图片/动图;视频只发封面 |
|
||||
| `site/mod.rs` 加 `MAX_MEDIA_UPLOAD_BYTES` 常量 | 不加 | 没有视频尺寸决策就不需要该常量,避免跨 crate 耦合 |
|
||||
| `b23.tv` 短链(跟随重定向) | 不匹配 | 多数短链指向视频,匹配后会把"静默忽略"变成用户的 "Failed to fetch media" |
|
||||
| `validate()` 校验 cookie | 不做 | 匿名可用,cookie 失效不致命;校验要额外请求一个端点,收益低 |
|
||||
| `media_headers` 给 hdslb 加 Referer | 返回 `None` | 实测图片与 durl 均无需 Referer(注释里记了这条验证) |
|
||||
| 计划阶段认为设备 cookie 是 YAGNI,不实现 | **实现**(`buvid3`+`buvid4`) | 计划之后做了对照实验:同一 IP 上"无 cookie → -352、只有 buvid3 → -352、buvid3+buvid4 → code:0",说明这是对本适配器主要失败模式的直接修复,而不是冗余保险 |
|
||||
| 只用不带参数的 `v1/detail` | 加 `features=itemOpusStyle` | 用户实测反馈"有内容的动态没有 title":不带参数时 opus 图文帖返回 legacy 形态,`desc` 为 `null`,正文与标题整个丢失。带参数后同一 ID 返回 `major.opus.summary.text` / `title` / `pics`。AV / 转发帖不受影响,legacy 形态仍保留为兜底 |
|
||||
|
||||
---
|
||||
|
||||
## 4. 测试与验证
|
||||
|
||||
- 单元(13):正则匹配/拒绝/忽略短链、缓存键归一、图片映射(https 归一 + 缩略图 + `.gif → Animated`)、
|
||||
封面、转发取 `orig` 媒体与正文拼接、纯文字无媒体、caption 转义、业务 code 分类(可重试性)、URL 归一、
|
||||
设备 cookie 拼装。
|
||||
- live(3,`#[ignore = "live network: …"]`):设备 cookie 可取、图片动态 2 图、纯文字动态无媒体。
|
||||
CI 的 `live` job 已覆盖。动态接口被风控时这两条 live 测试打印 `skipping:` 并提前返回(与 pixiv 的
|
||||
token 门控同款约定),设备 cookie 那条仍会真实执行。
|
||||
- 实测命令:
|
||||
`cargo run -p x-media --example fetch -- https://www.bilibili.com/opus/1245284537985925159`
|
||||
(输出 2 张 `https://i0.hdslb.com/…jpg` + `@518w.jpg` 缩略图 + 话题 tags)。
|
||||
- 全套:`cargo fmt --check`、`cargo clippy --workspace --all-targets -- -D warnings`、`cargo test --workspace` 全绿。
|
||||
|
||||
## 5. 已知限制
|
||||
|
||||
- 风控按 IP/请求量漂移,阶梯见 §2 最后一行:轻度靠设备 cookie 自愈,重度需 `BILIBILI_COOKIE` 或换 IP。
|
||||
被拦时按**可重试**失败处理(队列退避)+ 一条 warn,不会静默丢帖。
|
||||
- 接口 schema 会漂移(`module_dynamic.major` 实测可为 `null` 而正文留在 `desc`);DTO 全 `Option`,
|
||||
未知形态降级为"无媒体",不 panic。
|
||||
- 动态内嵌视频只发封面图(与 bff 同策略),不下载流。
|
||||
- 纯文字动态复用既有 "No media found" 回复。
|
||||
- `b23.tv` 短链不被匹配(见上表)。
|
||||
Generated
+602
-1154
File diff suppressed because it is too large
Load Diff
@@ -1,3 +1,12 @@
|
||||
[workspace]
|
||||
members = ["crates/x-media", "crates/xmedia-bot"]
|
||||
resolver = "3"
|
||||
|
||||
# Smaller/faster production binary: strip debug symbols, link-time
|
||||
# optimization across crates, and one codegen unit per crate (bigger LTO
|
||||
# wins). panic=abort is intentionally NOT set: queue workers and db
|
||||
# closures rely on JoinHandle catching panics, which abort would defeat.
|
||||
[profile.release]
|
||||
strip = true
|
||||
lto = "thin"
|
||||
codegen-units = 1
|
||||
|
||||
+21
-11
@@ -11,6 +11,13 @@ ARG APP_NAME=telegram-twitter-media-bot
|
||||
# runners. `/redirect/latest/` floats to the newest release build; each build
|
||||
# also ships a .sha256. Swap `amd64` for `arm64` when building arm64 images.
|
||||
ARG FFMPEG_URL=https://ffmpeg.martin-riedl.de/redirect/latest/linux/amd64/release/ffmpeg.zip
|
||||
# Arm64 images need this URL swapped for the `linux/arm64` build (currently
|
||||
# hardcoded amd64; the workflow builds amd64 only — see docker.yml).
|
||||
# Optional sha256 of ffmpeg.zip (pinned releases only): set to verify the
|
||||
# download. The mirror publishes .sha256 sidecars next to pinned builds, e.g.
|
||||
# https://ffmpeg.martin-riedl.de/download/linux/amd64/<id>_9.0/ffmpeg.zip.sha256
|
||||
# (the /redirect/latest/ URL itself has no sidecar — pin the effective URL).
|
||||
ARG FFMPEG_SHA256=
|
||||
|
||||
WORKDIR /build
|
||||
|
||||
@@ -23,27 +30,30 @@ COPY crates/xmedia-bot/Cargo.toml crates/xmedia-bot/Cargo.toml
|
||||
RUN mkdir -p crates/x-media/src crates/xmedia-bot/src \
|
||||
&& printf 'fn main() {}\n' > crates/xmedia-bot/src/main.rs \
|
||||
&& : > crates/x-media/src/lib.rs \
|
||||
&& cargo build --release -p xmedia-bot
|
||||
&& cargo build --release --locked -p xmedia-bot
|
||||
|
||||
# 2. Static ffmpeg next (cached unless FFMPEG_URL changes), so source edits
|
||||
# never re-download it. The zip contains a single `ffmpeg` binary at the
|
||||
# root. `unzip -t` verifies the archive before extraction so a bad
|
||||
# download fails loudly here instead of a cryptic later error.
|
||||
RUN wget -q -O /tmp/ffmpeg.zip "$FFMPEG_URL" \
|
||||
&& if [ -n "$FFMPEG_SHA256" ]; then echo "$FFMPEG_SHA256 /tmp/ffmpeg.zip" | sha256sum -c -; fi \
|
||||
&& unzip -tq /tmp/ffmpeg.zip \
|
||||
&& unzip -q /tmp/ffmpeg.zip -d /usr/local/bin \
|
||||
&& chmod +x /usr/local/bin/ffmpeg \
|
||||
&& rm /tmp/ffmpeg.zip \
|
||||
&& /usr/local/bin/ffmpeg -version >/dev/null
|
||||
|
||||
# 3. Real sources last: only our crates recompile on source changes. The
|
||||
# COPY preserves host mtimes, which predate the stub artifacts from step 1;
|
||||
# cargo's mtime-based freshness check would otherwise treat the stub build
|
||||
# as up-to-date and never compile the real sources. `touch` forces cargo to
|
||||
# see the real files as newer.
|
||||
# 3. Real sources last: only our crates recompile on source changes. Cargo's
|
||||
# freshness check is mtime-based; the COPY'd host files usually predate the
|
||||
# step-1 stub build, so cargo would consider the stub up to date and never
|
||||
# compile the real sources. `touch` makes every .rs newer than the stub
|
||||
# artifacts, forcing a rebuild of just the two crates while the compiled
|
||||
# dependency layer stays cached. (`cargo clean -p` does NOT work here — it
|
||||
# removes 0 files and the stub binary silently ships.)
|
||||
COPY crates/ ./crates/
|
||||
RUN find crates -type f -name '*.rs' -exec touch {} + \
|
||||
&& cargo build --release -p xmedia-bot
|
||||
&& cargo build --release --locked -p xmedia-bot
|
||||
|
||||
# ---------- runtime stage ----------
|
||||
FROM debian:bookworm-slim
|
||||
@@ -56,10 +66,10 @@ LABEL org.opencontainers.image.title="${APP_NAME}"
|
||||
|
||||
# Everything is copied in — no apt in the runtime stage. Privilege dropping is
|
||||
# done by docker-entrypoint.sh with setpriv (util-linux, already in
|
||||
# bookworm-slim), so no gosu needed.
|
||||
COPY --from=builder /etc/ssl/certs/ca-certificates.crt /etc/ssl/certs/ca-certificates.crt
|
||||
COPY --from=builder /usr/lib/x86_64-linux-gnu/libssl.so.3* /usr/lib/x86_64-linux-gnu/
|
||||
COPY --from=builder /usr/lib/x86_64-linux-gnu/libcrypto.so.3* /usr/lib/x86_64-linux-gnu/
|
||||
# bookworm-slim), so no gosu needed. TLS is rustls (webpki-roots baked in,
|
||||
# see Cargo.toml feature `rustls`/`rustls-tls`), so no system CA bundle or
|
||||
# libssl are needed; the static ffmpeg only processes local files (all
|
||||
# downloads go through reqwest).
|
||||
COPY --from=builder /usr/local/bin/ffmpeg /usr/local/bin/ffmpeg
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
+131
@@ -0,0 +1,131 @@
|
||||
# TelegramXMediaBot
|
||||
|
||||
A Telegram bot that turns post links from X / Twitter, Pixiv, Bluesky, Misskey (misskey.io), and Bilibili dynamics into media messages (images, video, GIF) with the post's title, author, and tags.
|
||||
|
||||
## Features
|
||||
|
||||
- Sending a link in a private chat fetches and sends the images, videos and GIFs automatically; oversized media is split into batches
|
||||
- Text-only posts report "no media"; unsupported links are silently ignored
|
||||
- Long posts (text ≥ `CAPTION_QUOTE_TEXT_CHARS`, default 200) show **the text part** of their caption inside a collapsible blockquote, with the link and author line left outside it
|
||||
- Inline queries (`@bot <link>`)
|
||||
- Bind a forward channel for automatic forwarding; edit the caption before forwarding and apply custom templates
|
||||
- Failed sends are retried automatically with persistence; the user is notified after retries are exhausted
|
||||
- Pixiv ugoira animations are transcoded to MP4; Bluesky videos are remuxed (HLS stream → MP4)
|
||||
- Photos exceeding Telegram's size/dimension limits are compressed automatically (original format kept, JPEG fallback only when needed)
|
||||
- Link-result cache: after a successful send the Telegram file ids and caption fields are cached locally, so a repeated link is re-sent from local state — no source-site request, no media file stored (expiry controlled by `LINK_CACHE_TTL_SECONDS`, default 7 days)
|
||||
|
||||
## Quick start
|
||||
|
||||
```bash
|
||||
# Required: BotFather token; optional: PIXIV_REFRESH_TOKEN (Pixiv is disabled without it)
|
||||
export TELOXIDE_TOKEN=<token>
|
||||
export PIXIV_REFRESH_TOKEN=<token>
|
||||
|
||||
cargo run -p xmedia-bot
|
||||
```
|
||||
|
||||
Docker deployment (see `docker-compose.yml.example`):
|
||||
|
||||
```bash
|
||||
docker build -t tgxmb .
|
||||
docker run --rm -d --name tgxmb --env-file .env -v ./data:/app/data tgxmb
|
||||
```
|
||||
|
||||
Environment variables: `TELOXIDE_TOKEN` (required), `PIXIV_REFRESH_TOKEN`, `BOT_ADMIN`, `EDIT_MESSAGE_TTL_SECONDS`, `LINK_CACHE_TTL_SECONDS`, `RUST_LOG`, `TELOXIDE_PROXY`, `WEBHOOK*`, `TWITTER_AUTH_TOKEN` (optional), `BILIBILI_COOKIE` (optional).
|
||||
|
||||
NSFW tweets: the public syndication endpoint does not return sensitive content. Setting `TWITTER_AUTH_TOKEN` (the `auth_token` cookie value of a logged-in x.com session) lets the bot fetch NSFW media in the logged-in state only when it hits a withheld tweet; without it, the bot reports no media.
|
||||
|
||||
Bilibili dynamics are fetched anonymously by default (no login; the bot fetches bilibili's anonymous `buvid3`/`buvid4` device cookies itself to raise the success rate). If the server's egress IP gets hard-flagged by bilibili (persistent `risk control (-352)` log lines or HTTP 412), set `BILIBILI_COOKIE` (the whole cookie string from a logged-in browser, e.g. `SESSDATA=…; bili_jct=…`) to restore access. Only a dynamic's images and animations are sent; an attached video degrades to its cover image.
|
||||
|
||||
### Webhook deployment (needs a reverse proxy)
|
||||
|
||||
`docker-compose.yml.example` ships an [nginx-proxy](https://github.com/nginx-proxy/nginx-proxy) + [acme-companion](https://github.com/nginx-proxy/acme-companion) reverse-proxy orchestration. Pick one deployment shape:
|
||||
|
||||
**With a domain**
|
||||
1. Point a DNS A record at the server
|
||||
2. In compose set `VIRTUAL_HOST` and `WEBHOOK_URL` to the domain, and uncomment `ACME_HOST` (set it to the domain)
|
||||
3. acme-companion issues and renews certificates automatically — nothing manual
|
||||
|
||||
**IP only**
|
||||
Let's Encrypt can issue certificates for public IPs (available since 2026, validity ~7 days, requires the `shortlived` profile). Use [acme.sh](https://github.com/acmesh-official/acme.sh) to issue and renew automatically, no manual certificates:
|
||||
|
||||
1. Add an acme-ip service to compose (issue + daily auto-renewal check):
|
||||
```yaml
|
||||
acme-ip:
|
||||
image: neilpang/acme.sh
|
||||
container_name: acme-ip
|
||||
command: daemon
|
||||
restart: always
|
||||
volumes:
|
||||
- certs:/acme.sh
|
||||
- html:/usr/share/nginx/html
|
||||
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||
networks: [proxy]
|
||||
```
|
||||
2. First issuance (replace `<SERVER_IP>` with the server's public IP; IPv6 works too, repeat `-d` for more):
|
||||
```bash
|
||||
docker compose exec acme-ip acme.sh --issue --server letsencrypt \
|
||||
-d <SERVER_IP> --cert-profile shortlived --days 3 \
|
||||
--webroot /usr/share/nginx/html \
|
||||
--install-cert --cert-file /acme.sh/<SERVER_IP>.crt \
|
||||
--key-file /acme.sh/<SERVER_IP>.key \
|
||||
--reloadcmd "curl --unix-socket /var/run/docker.sock -X POST http://localhost/containers/nginx-proxy/kill?signal=HUP"
|
||||
```
|
||||
3. In compose set `VIRTUAL_HOST: '<SERVER_IP>'` and `WEBHOOK_URL: 'https://<SERVER_IP>/'`; no `WEBHOOK_CERT` needed. Renewal is handled by the acme.sh daemon (`--days 3` = renew every 3 days, buffer against the 7-day validity), and a successful renewal HUP-notifies nginx-proxy to load the new certificate.
|
||||
|
||||
Limitations: certificate validity ~7 days; only http-01/tls-alpn-01 validation (port 80 must be publicly reachable); no DNS-01, private IPs or IP ranges; at most 5 certificates per 168 hours for the same IP set. It is recommended to trial-issue with `--server letsencrypt_test` first, then switch to the production server.
|
||||
|
||||
Telegram only accepts ports 443/80/88/8443.
|
||||
|
||||
<details>
|
||||
<summary>Environment variables</summary>
|
||||
|
||||
| Variable | Description |
|
||||
|---|---|
|
||||
| `TELOXIDE_TOKEN` | Bot token (required) |
|
||||
| `PIXIV_REFRESH_TOKEN` | Pixiv refresh token; Pixiv is disabled without it |
|
||||
| `BILIBILI_COOKIE` | Optional bilibili cookie string (`SESSDATA=…; bili_jct=…`); only needed when the egress IP stays risk-controlled (device cookies are fetched automatically) |
|
||||
| `BOT_ADMIN` | Admin chat IDs, comma-separated; receives start/stop notifications |
|
||||
| `EDIT_MESSAGE_TTL_SECONDS` | Edit-before-forward record expiry in seconds, default 86400 |
|
||||
| `LINK_CACHE_TTL_SECONDS` | Link-result cache expiry in seconds, default 604800 (7 days) |
|
||||
| `CAPTION_QUOTE_TEXT_CHARS` | **The text part** of the caption (the joined `{title}` + `{content}`) is wrapped in a collapsible blockquote once it reaches this many characters, default 200; `0` disables |
|
||||
| `DATA_DIR` | Data directory (where the SQLite `task_queue.db` lives), default `data` (relative to the working directory, created automatically) |
|
||||
| `RUST_LOG` | Log level |
|
||||
| `TELOXIDE_PROXY` | HTTP proxy (e.g. `http://127.0.0.1:10808`); applies to both the Telegram Bot API and site fetches — required on restricted networks (e.g. behind the GFW) |
|
||||
| `LOCAL_USER_ID` | UID the container runs as, default 9001 |
|
||||
| `VIRTUAL_HOST` | Public domain or IP; nginx-proxy routes by this |
|
||||
| `VIRTUAL_PORT` | Port the bot listens on inside the container; nginx-proxy's forwarding target |
|
||||
| `ACME_HOST` | Domain deployment: when set to the domain, acme-companion issues/renews certificates automatically |
|
||||
| `DEFAULT_HOST` | nginx-proxy routes requests with unknown Host headers to this vhost (needed for IP access) |
|
||||
| `DEFAULT_EMAIL` | acme-companion certificate notification email |
|
||||
| `WEBHOOK` | `true` enables webhook mode (polling by default) |
|
||||
| `WEBHOOK_LISTEN` / `WEBHOOK_PORT` | Listen address/port inside the bot container |
|
||||
| `WEBHOOK_URL` | Public HTTPS URL (`https://domain/` or `https://IP/`) |
|
||||
| `WEBHOOK_CERT` | Optional; self-signed certificate path, only used for Telegram-side validation (TLS is terminated by the reverse proxy) |
|
||||
| `WEBHOOK_SECRET_TOKEN` | Update validation token (`X-Telegram-Bot-Api-Secret-Token`) |
|
||||
|
||||
</details>
|
||||
|
||||
## Commands
|
||||
|
||||
| Command | Description |
|
||||
|---|---|
|
||||
| `/start` | Welcome message |
|
||||
| `/help` | List all commands and usage (this command table) |
|
||||
| `/set_forward_channel <channel>` | Set the forward channel: `@channel` or channel ID; media messages are forwarded to it automatically afterwards |
|
||||
| `/remove_forward_channel` | Remove the forward channel |
|
||||
| `/edit_before_forward` | Toggle "edit before forward": when enabled, the bot posts a prompt after forwarding; replying to it edits the first forwarded message's caption (or taps a template button to apply one) |
|
||||
| `/set_template <name>` | Reply to a message containing `[]` to save it as a named template; `[]` is replaced by the original post link when forwarding (used with "edit before forward") |
|
||||
| `/set_format <site> <format>` | Customize the caption format for one site. Sites: `twitter` / `bsky` / `pixiv` / `misskey` / `bilibili`. Placeholders: `{url}` `{author}` `{author_url}` `{title}` `{content}` `{tags}` |
|
||||
| `/clear_cache [link]` | Clear the link cache (admin only); with a link only that entry, otherwise everything |
|
||||
| `/bot_dict` | Show the current chat state (debugging; admin only) |
|
||||
| `/test <link>` | Parse a link and send its media; no channel forward, no edit-before-forward prompt (send only) |
|
||||
| `/debug <link>` | Debug: parse a link and report the parse result only (site, title, author, tags, media list) — no media is sent |
|
||||
|
||||
Link processing works only in private chats; commands work in any chat.
|
||||
|
||||
## Notes
|
||||
|
||||
- State is persisted in `data/task_queue.db`; compose deployments use the bind mount `./data` (keep it a directory for easy backups)
|
||||
- The runtime needs ffmpeg (built into the Docker image)
|
||||
- Tests: `cargo test --workspace`
|
||||
@@ -1,15 +1,17 @@
|
||||
# TelegramXMediaBot
|
||||
|
||||
Telegram 机器人,将 X / Twitter、Pixiv、Bluesky 的帖子链接转换为媒体消息发送,附带帖子标题、作者与标签。
|
||||
Telegram 机器人,将 X / Twitter、Pixiv、Bluesky、Misskey (misskey.io)、Bilibili 动态的帖子链接转换为媒体消息发送,附带帖子标题、作者与标签。
|
||||
|
||||
## 功能
|
||||
|
||||
- 私聊发送链接后自动抓取并发送图片、视频与 GIF,超量图片自动分批
|
||||
- 纯文字帖提示无媒体;不支持的链接静默忽略
|
||||
- 长帖(正文 ≥ `CAPTION_QUOTE_TEXT_CHARS`,默认 200)的**正文部分**用可折叠引用块展示,链接与作者行留在引用块外
|
||||
- 支持内联查询(`@机器人 <链接>`)
|
||||
- 可绑定转发频道自动转发;支持转发前编辑 caption 与自定义模板
|
||||
- 发送失败自动重试并持久化,重试耗尽后通知用户
|
||||
- Pixiv ugoira 动图自动转码为 MP4
|
||||
- Pixiv ugoira 动图自动转码为 MP4;Bluesky 视频自动转码(HLS 流 → MP4)
|
||||
- 超过 Telegram 尺寸/大小限制的图片自动压缩(保持原格式,必要时转 JPEG)
|
||||
- 链接结果本地缓存:成功发送后缓存 Telegram file id 与 caption 等,再次收到相同链接直接本地重发,不再请求源站、不保存媒体文件(`LINK_CACHE_TTL_SECONDS` 控制过期,默认 7 天)
|
||||
|
||||
## 快速开始
|
||||
@@ -29,10 +31,12 @@ docker build -t tgxmb .
|
||||
docker run --rm -d --name tgxmb --env-file .env -v ./data:/app/data tgxmb
|
||||
```
|
||||
|
||||
环境变量:`TELOXIDE_TOKEN`(必填)、`PIXIV_REFRESH_TOKEN`、`BOT_ADMIN`、`EDIT_MESSAGE_TTL_SECONDS`、`LINK_CACHE_TTL_SECONDS`、`RUST_LOG`、`WEBHOOK*`、`TWITTER_AUTH_TOKEN`(可选)。
|
||||
环境变量:`TELOXIDE_TOKEN`(必填)、`PIXIV_REFRESH_TOKEN`、`BOT_ADMIN`、`EDIT_MESSAGE_TTL_SECONDS`、`LINK_CACHE_TTL_SECONDS`、`RUST_LOG`、`TELOXIDE_PROXY`、`WEBHOOK*`、`TWITTER_AUTH_TOKEN`(可选)、`BILIBILI_COOKIE`(可选)。
|
||||
|
||||
NSFW 推文:公开的 syndication 接口不返回敏感内容。设置 `TWITTER_AUTH_TOKEN`(登录 x.com 后浏览器 Cookie 里的 `auth_token` 值)后,bot 会仅在遇到 NSFW 推文时以登录态获取媒体;未设置则提示无媒体。
|
||||
|
||||
Bilibili 动态默认匿名抓取(无需登录,bot 会自动从 B 站的匿名指纹接口取 `buvid3`/`buvid4` 设备 cookie 以提高成功率)。若服务器出口 IP 被 B 站重度风控(日志里的 `risk control (-352)` 或 HTTP 412,且持续出现),设置 `BILIBILI_COOKIE`(登录后浏览器里整条 Cookie 串,如 `SESSDATA=…; bili_jct=…`)可恢复访问。当前只发送动态里的图片与动图,动态内嵌视频发送其封面。
|
||||
|
||||
### Webhook 部署(需要反向代理)
|
||||
|
||||
`docker-compose.yml.example` 内置了 [nginx-proxy](https://github.com/nginx-proxy/nginx-proxy) + [acme-companion](https://github.com/nginx-proxy/acme-companion) 反向代理编排,按部署环境二选一:
|
||||
@@ -80,10 +84,14 @@ Telegram 只接受 443/80/88/8443 端口。
|
||||
|---|---|
|
||||
| `TELOXIDE_TOKEN` | Bot token(必填) |
|
||||
| `PIXIV_REFRESH_TOKEN` | Pixiv 刷新令牌;未设置则禁用 Pixiv |
|
||||
| `BILIBILI_COOKIE` | 可选的 B 站 Cookie 串(`SESSDATA=…; bili_jct=…`),仅在出口 IP 被持续风控时才需要(设备 cookie 由 bot 自动获取) |
|
||||
| `BOT_ADMIN` | 管理员聊天 ID,逗号分隔;接收启动/停止通知 |
|
||||
| `EDIT_MESSAGE_TTL_SECONDS` | 转发前编辑记录过期秒数,默认 86400 |
|
||||
| `LINK_CACHE_TTL_SECONDS` | 链接结果缓存过期秒数,默认 604800(7 天) |
|
||||
| `CAPTION_QUOTE_TEXT_CHARS` | 正文(`{title}` + `{content}` 合计)达到该长度(字符)时,caption 的**正文部分**用可折叠引用块包裹,默认 200;`0` 关闭 |
|
||||
| `DATA_DIR` | 数据目录(SQLite 数据库 `task_queue.db` 所在目录),默认 `data`(相对工作目录,会自动创建) |
|
||||
| `RUST_LOG` | 日志级别 |
|
||||
| `TELOXIDE_PROXY` | HTTP 代理(如 `http://127.0.0.1:10808`);同时作用于 Telegram Bot API 与站点抓取请求,网络受限环境(如 GFW)必需 |
|
||||
| `LOCAL_USER_ID` | 容器内运行用户 UID,默认 9001 |
|
||||
| `VIRTUAL_HOST` | 对外域名或 IP,nginx-proxy 按此路由 |
|
||||
| `VIRTUAL_PORT` | bot 容器内监听端口,nginx-proxy 的转发目标 |
|
||||
@@ -93,6 +101,7 @@ Telegram 只接受 443/80/88/8443 端口。
|
||||
| `WEBHOOK` | `true` 启用 webhook 模式(默认轮询) |
|
||||
| `WEBHOOK_LISTEN` / `WEBHOOK_PORT` | bot 容器内监听地址/端口 |
|
||||
| `WEBHOOK_URL` | 对外公网 HTTPS 地址(`https://域名/` 或 `https://IP/`) |
|
||||
| `WEBHOOK_CERT` | 可选;自签名证书路径,仅用于 Telegram 侧验证(TLS 由反向代理终止) |
|
||||
| `WEBHOOK_SECRET_TOKEN` | 更新校验令牌(`X-Telegram-Bot-Api-Secret-Token`) |
|
||||
|
||||
</details>
|
||||
@@ -107,8 +116,11 @@ Telegram 只接受 443/80/88/8443 端口。
|
||||
| `/remove_forward_channel` | 取消转发频道 |
|
||||
| `/edit_before_forward` | 开关「转发前编辑」:开启后,转发成功后 bot 会发一条提示消息,回复它可修改第一条转发消息的 caption(或点击模板按钮套用模板) |
|
||||
| `/set_template <名称>` | 回复一条含 `[]` 的消息,将其保存为命名模板;转发时 `[]` 会被替换为原帖链接(配合「转发前编辑」使用) |
|
||||
| `/set_format <站点> <格式>` | 自定义某站点的 caption 格式。站点:`twitter` / `bsky` / `pixiv`。占位符:`{url}` `{author}` `{author_url}` `{title}` `{tags}` |
|
||||
| `/bot_dict` | 查看当前聊天状态(调试用) |
|
||||
| `/set_format <站点> <格式>` | 自定义某站点的 caption 格式。站点:`twitter` / `bsky` / `pixiv` / `misskey` / `bilibili`。占位符:`{url}` `{author}` `{author_url}` `{title}` `{content}` `{tags}` |
|
||||
| `/clear_cache [链接]` | 清空链接缓存(仅管理员);带链接只清该条,否则清空全部 |
|
||||
| `/bot_dict` | 查看当前聊天状态(调试用;仅管理员) |
|
||||
| `/test <链接>` | 解析链接并发送媒体;不转发到频道、不弹转发前编辑提示(仅发送) |
|
||||
| `/debug <链接>` | 调试:只解析链接并返回解析结果(站点、标题、作者、标签、媒体列表),不发送任何媒体 |
|
||||
|
||||
链接处理仅限私聊;命令在任意聊天可用。
|
||||
|
||||
|
||||
@@ -1,19 +1,20 @@
|
||||
[package]
|
||||
name = "x-media"
|
||||
version = "1.0.8"
|
||||
version = "1.7.0"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
reqwest = { version = "0.13", features = ["json", "query", "form"] }
|
||||
reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls"] }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
regex = "1.12"
|
||||
html-escape = "0.2"
|
||||
url = "2.5.2"
|
||||
bytes = "1"
|
||||
zip = "2"
|
||||
zip = "8"
|
||||
tempfile = "3"
|
||||
rand = "0.8"
|
||||
thiserror = "2"
|
||||
rand = "0.10"
|
||||
log = "0.4"
|
||||
tokio = { version = "1.40", features = ["time"] }
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,6 @@
|
||||
mod interface;
|
||||
mod model;
|
||||
|
||||
pub use interface::{
|
||||
BilibiliSite, PATTERN, cache_key, enabled, fetch_from_url, is_retryable, media_headers,
|
||||
};
|
||||
@@ -0,0 +1,158 @@
|
||||
//! Serde DTOs for the Bilibili dynamic detail endpoint
|
||||
//! (`/x/polymer/web-dynamic/v1/detail`), mirroring live responses
|
||||
//! (field paths verified 2026-09-17). Every field is optional so an API
|
||||
//! shape change degrades to "no media" instead of a parse failure.
|
||||
|
||||
use serde::Deserialize;
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Detail {
|
||||
/// Business code: `0` = OK, `-352`/`-412` = risk control, `500`/`4101147`
|
||||
/// = gone.
|
||||
pub(crate) code: i64,
|
||||
#[serde(default)]
|
||||
pub(crate) message: Option<String>,
|
||||
#[serde(default)]
|
||||
pub(crate) data: Option<Data>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Data {
|
||||
#[serde(default)]
|
||||
pub(crate) item: Option<Box<Item>>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Item {
|
||||
/// The dynamic id, same numeric id as in the URL.
|
||||
#[serde(default)]
|
||||
pub(crate) id_str: String,
|
||||
#[serde(default)]
|
||||
pub(crate) modules: Option<Modules>,
|
||||
/// The quoted dynamic when this item is a forward. A forward shell often
|
||||
/// carries no media of its own — the original holds it.
|
||||
#[serde(default)]
|
||||
pub(crate) orig: Option<Box<Item>>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Modules {
|
||||
#[serde(default)]
|
||||
pub(crate) module_author: Option<Author>,
|
||||
#[serde(default)]
|
||||
pub(crate) module_dynamic: Option<Dynamic>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Author {
|
||||
#[serde(default)]
|
||||
pub(crate) name: String,
|
||||
#[serde(default)]
|
||||
pub(crate) mid: Option<i64>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Dynamic {
|
||||
#[serde(default)]
|
||||
pub(crate) desc: Option<Desc>,
|
||||
#[serde(default)]
|
||||
pub(crate) major: Option<Major>,
|
||||
/// A single topic (`{"id":…,"name":…}`), the dynamic's only tag source.
|
||||
#[serde(default)]
|
||||
pub(crate) topic: Option<Topic>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Desc {
|
||||
#[serde(default)]
|
||||
pub(crate) text: String,
|
||||
}
|
||||
|
||||
/// `major` is a tagged union: `type` (`MAJOR_TYPE_DRAW` / `_OPUS` /
|
||||
/// `_ARCHIVE` / …) plus one payload object per type. Only the three payloads
|
||||
/// this adapter reads are modeled; an unknown major simply yields no media.
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Major {
|
||||
#[serde(default)]
|
||||
pub(crate) draw: Option<Draw>,
|
||||
#[serde(default)]
|
||||
pub(crate) opus: Option<Opus>,
|
||||
#[serde(default)]
|
||||
pub(crate) archive: Option<Archive>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Draw {
|
||||
#[serde(default)]
|
||||
pub(crate) items: Vec<Pic>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Pic {
|
||||
/// `major.draw` image URL.
|
||||
#[serde(default)]
|
||||
pub(crate) src: Option<String>,
|
||||
/// `major.opus.pics` image URL — the opus shape names the field
|
||||
/// differently while carrying the same image.
|
||||
#[serde(default)]
|
||||
pub(crate) url: Option<String>,
|
||||
}
|
||||
|
||||
impl Pic {
|
||||
/// The image URL, whichever key this serialization put it under.
|
||||
pub(crate) fn url(&self) -> Option<&str> {
|
||||
self.src.as_deref().or(self.url.as_deref())
|
||||
}
|
||||
}
|
||||
|
||||
/// `major.opus`: the serialization of an image/text post the web client asks
|
||||
/// for (`features=itemOpusStyle`). It carries the parts the legacy shape drops
|
||||
/// entirely — the document title and body of an opus post, whose
|
||||
/// `module_dynamic.desc` comes back `null`.
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Opus {
|
||||
/// Document headline; often absent.
|
||||
#[serde(default)]
|
||||
pub(crate) title: Option<String>,
|
||||
/// Document body (untruncated: a 307-char sample came back whole).
|
||||
#[serde(default)]
|
||||
pub(crate) summary: Option<Desc>,
|
||||
#[serde(default)]
|
||||
pub(crate) pics: Vec<Pic>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Archive {
|
||||
/// The attached video's cover — the only image an AV dynamic has (the
|
||||
/// video itself is deliberately not resolved, see the module docs).
|
||||
#[serde(default)]
|
||||
pub(crate) cover: Option<String>,
|
||||
/// The video's title. An AV dynamic has no body of its own (`desc` comes
|
||||
/// back `null`), so this card title is the post's content.
|
||||
#[serde(default)]
|
||||
pub(crate) title: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Topic {
|
||||
#[serde(default)]
|
||||
pub(crate) name: String,
|
||||
}
|
||||
|
||||
/// Response of the anonymous fingerprint endpoint (`/x/frontend/finger/spi`),
|
||||
/// the source of the adapter's device cookies.
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Fingerprint {
|
||||
#[serde(default)]
|
||||
pub(crate) data: Option<FingerprintData>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct FingerprintData {
|
||||
/// Sent as the `buvid3` cookie.
|
||||
#[serde(default, rename = "b_3")]
|
||||
pub(crate) buvid3: String,
|
||||
/// Sent as the `buvid4` cookie.
|
||||
#[serde(default, rename = "b_4")]
|
||||
pub(crate) buvid4: String,
|
||||
}
|
||||
@@ -1,12 +1,34 @@
|
||||
use super::model;
|
||||
use crate::media::Media;
|
||||
use crate::site::{FetchError, Fetched};
|
||||
use html_escape::encode_text;
|
||||
use crate::site::{FetchError, Fetched, Site, SiteFuture};
|
||||
use html_escape::{encode_double_quoted_attribute, encode_text};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static PATTERN: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"bsky\.app/profile/([\w.\-:]+)/post/([\w.\-~]+)").unwrap());
|
||||
/// Registry entry for the bluesky adapter (see [`crate::site::Site`]).
|
||||
pub struct BskySite;
|
||||
|
||||
impl Site for BskySite {
|
||||
fn id(&self) -> &'static str {
|
||||
"bsky"
|
||||
}
|
||||
|
||||
fn pattern(&self) -> &'static Regex {
|
||||
&PATTERN
|
||||
}
|
||||
|
||||
fn cache_key(&self, url: &str) -> Option<String> {
|
||||
cache_key(url)
|
||||
}
|
||||
|
||||
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched> {
|
||||
Box::pin(async move { fetch_from_url(url).await })
|
||||
}
|
||||
}
|
||||
|
||||
pub static PATTERN: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"^(?:https?://)?bsky\.app/profile/([\w.\-:]+)/post/([\w.\-~]+)").unwrap()
|
||||
});
|
||||
|
||||
pub fn enabled() -> bool {
|
||||
true
|
||||
@@ -22,7 +44,180 @@ pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> {
|
||||
.get(2)
|
||||
.map(|m| m.as_str())
|
||||
.ok_or(FetchError::NotFound)?;
|
||||
Ok(fetch(handle, rkey).await?.into())
|
||||
let post = fetch(handle, rkey).await?;
|
||||
let mut fetched: Fetched = post.into();
|
||||
// bsky video embeds expose only an HLS playlist URL, which Telegram
|
||||
// cannot fetch; remux it to a single MP4 (mirrors the pixiv ugoira
|
||||
// encode path — the temp file stays alive via `_keep_alive`). On any
|
||||
// failure the video item is dropped and the post degrades to its text.
|
||||
let mut media = Vec::with_capacity(fetched.media.len());
|
||||
for item in fetched.media {
|
||||
let is_hls = matches!(&item, Media::Video { url, .. }
|
||||
if url.contains("playlist") || url.ends_with(".m3u8"));
|
||||
if !is_hls {
|
||||
media.push(item);
|
||||
continue;
|
||||
}
|
||||
let url = item.url().to_string();
|
||||
match resolve_bsky_video(&url).await {
|
||||
Ok(Some((mp4_path, keep_alive))) => {
|
||||
let thumbnail_url = match &item {
|
||||
Media::Video { thumbnail_url, .. } => thumbnail_url.clone(),
|
||||
_ => String::new(),
|
||||
};
|
||||
media.push(Media::Video {
|
||||
title: None,
|
||||
url: mp4_path.to_string_lossy().into_owned(),
|
||||
thumbnail_url,
|
||||
});
|
||||
fetched._keep_alive = Some(keep_alive);
|
||||
}
|
||||
Ok(None) => log::warn!("bsky video remux unavailable for {url}"),
|
||||
Err(e) => log::warn!("bsky video remux failed for {url}: {e}"),
|
||||
}
|
||||
}
|
||||
fetched.media = media;
|
||||
Ok(fetched)
|
||||
}
|
||||
|
||||
/// Cache key for a bsky URL: `"bsky:<handle>/<rkey>"`. The prefix is the
|
||||
/// site id used for caption-format lookup and link-cache keys.
|
||||
pub fn cache_key(url: &str) -> Option<String> {
|
||||
PATTERN
|
||||
.captures(url)
|
||||
.map(|caps| format!("bsky:{}/{}", &caps[1], &caps[2]))
|
||||
}
|
||||
|
||||
/// Bluesky's fetch-retry policy: transient classes only. Not-found, blocked
|
||||
/// and parse failures are permanent.
|
||||
pub fn is_retryable(err: &FetchError) -> bool {
|
||||
matches!(err, FetchError::Http(_) | FetchError::Transient(_))
|
||||
}
|
||||
|
||||
/// bsky media (cdn.bsky.app) needs no extra headers.
|
||||
pub fn media_headers(_url: &str) -> Option<Vec<(&'static str, String)>> {
|
||||
None
|
||||
}
|
||||
|
||||
/// Downloads an HLS playlist (master or media) and remuxes its segments to a
|
||||
/// single MP4 via ffmpeg. Returns the MP4 path plus the temp dir that must
|
||||
/// stay alive until the file is uploaded. `Ok(None)` when ffmpeg is missing.
|
||||
///
|
||||
/// Verified live (2026-08): bsky master playlists carry `#EXT-X-STREAM-INF`
|
||||
/// variant lines (e.g. `720p/video.m3u8?session_id=…`), and the media
|
||||
/// playlists are VOD MPEG-TS segments (`videoN.ts?…`) without EXT-X-MAP, so
|
||||
/// a plain `-f concat -c copy` remux is valid.
|
||||
async fn resolve_bsky_video(
|
||||
playlist_url: &str,
|
||||
) -> Result<Option<(std::path::PathBuf, tempfile::TempDir)>, String> {
|
||||
if !crate::site::ffmpeg_available() {
|
||||
crate::site::log_once_ffmpeg_missing();
|
||||
return Ok(None);
|
||||
}
|
||||
let master = crate::site::download_media_limited(playlist_url, 1_048_576)
|
||||
.await
|
||||
.map_err(|e| format!("bsky video master playlist: {e}"))?;
|
||||
let master = String::from_utf8_lossy(&master);
|
||||
|
||||
// Master playlist: pick the variant with the highest declared bandwidth.
|
||||
let playlist_url = if master.contains("#EXT-X-STREAM-INF") {
|
||||
let mut best: Option<(u64, String)> = None;
|
||||
let mut lines = master.lines();
|
||||
while let Some(line) = lines.next() {
|
||||
if !line.starts_with("#EXT-X-STREAM-INF") {
|
||||
continue;
|
||||
}
|
||||
let bandwidth = line
|
||||
.split_once("BANDWIDTH=")
|
||||
.and_then(|(_, rest)| rest.split(|c: char| !c.is_ascii_digit()).next())
|
||||
.and_then(|n| n.parse::<u64>().ok())
|
||||
.unwrap_or(0);
|
||||
if let Some(uri) = lines.next().filter(|u| !u.starts_with('#'))
|
||||
&& bandwidth >= best.as_ref().map(|(b, _)| *b).unwrap_or(0)
|
||||
{
|
||||
best = Some((bandwidth, uri.to_string()));
|
||||
}
|
||||
}
|
||||
let Some((_, uri)) = best else {
|
||||
return Err("bsky video master playlist has no variants".to_string());
|
||||
};
|
||||
url::Url::parse(playlist_url)
|
||||
.and_then(|base| base.join(&uri))
|
||||
.map_err(|e| format!("bsky video variant URL: {e}"))?
|
||||
.to_string()
|
||||
} else {
|
||||
playlist_url.to_string()
|
||||
};
|
||||
|
||||
let variant = crate::site::download_media_limited(&playlist_url, 1_048_576)
|
||||
.await
|
||||
.map_err(|e| format!("bsky video media playlist: {e}"))?;
|
||||
let variant = String::from_utf8_lossy(&variant);
|
||||
// Segment URIs: non-#, non-empty lines, resolved relative to the playlist.
|
||||
let base = url::Url::parse(&playlist_url).map_err(|e| format!("bsky playlist URL: {e}"))?;
|
||||
let segments: Vec<String> = variant
|
||||
.lines()
|
||||
.map(str::trim)
|
||||
.filter(|l| !l.is_empty() && !l.starts_with('#'))
|
||||
.map(|l| base.join(l).map(|u| u.to_string()))
|
||||
.collect::<Result<_, _>>()
|
||||
.map_err(|e| format!("bsky segment URL: {e}"))?;
|
||||
if segments.is_empty() {
|
||||
return Err("bsky video playlist has no segments".to_string());
|
||||
}
|
||||
if segments.len() > 500 {
|
||||
return Err("bsky video has too many segments".to_string());
|
||||
}
|
||||
|
||||
let frames_dir = tempfile::tempdir().map_err(|e| e.to_string())?;
|
||||
let out_dir = tempfile::tempdir().map_err(|e| e.to_string())?;
|
||||
let mut total: u64 = 0;
|
||||
let mut list = String::new();
|
||||
for (i, seg) in segments.iter().enumerate() {
|
||||
let bytes = crate::site::download_media_limited(seg, 20 * 1024 * 1024)
|
||||
.await
|
||||
.map_err(|e| format!("bsky segment {i}: {e}"))?;
|
||||
total += bytes.len() as u64;
|
||||
if total > 256 * 1024 * 1024 {
|
||||
return Err("bsky video exceeds total size cap".to_string());
|
||||
}
|
||||
let path = frames_dir.path().join(format!("seg_{i:04}.ts"));
|
||||
std::fs::write(&path, &bytes).map_err(|e| e.to_string())?;
|
||||
list.push_str(&format!("file '{}'\n", path.to_string_lossy()));
|
||||
}
|
||||
let list_path = frames_dir.path().join("list.txt");
|
||||
std::fs::write(&list_path, &list).map_err(|e| e.to_string())?;
|
||||
|
||||
let output = out_dir.path().join("video.mp4");
|
||||
let list_str = list_path.to_string_lossy().into_owned();
|
||||
let output_str = output.to_string_lossy().into_owned();
|
||||
let status = tokio::task::spawn_blocking(move || {
|
||||
std::process::Command::new("ffmpeg")
|
||||
.args([
|
||||
"-y",
|
||||
"-f",
|
||||
"concat",
|
||||
"-safe",
|
||||
"0",
|
||||
"-i",
|
||||
&list_str,
|
||||
"-c",
|
||||
"copy",
|
||||
"-movflags",
|
||||
"+faststart",
|
||||
&output_str,
|
||||
])
|
||||
.stdout(std::process::Stdio::null())
|
||||
.stderr(std::process::Stdio::null())
|
||||
.status()
|
||||
})
|
||||
.await
|
||||
.map_err(|e| format!("bsky remux worker panicked: {e}"))?;
|
||||
match status {
|
||||
Ok(s) if s.success() => Ok(Some((output, out_dir))),
|
||||
Ok(s) => Err(format!("ffmpeg exited with {s}")),
|
||||
Err(e) => Err(format!("ffmpeg spawn failed: {e}")),
|
||||
}
|
||||
}
|
||||
|
||||
/// Fetches a post thread by handle or DID (`at://` URIs work for both).
|
||||
@@ -35,8 +230,16 @@ pub async fn fetch(handle: &str, rkey: &str) -> Result<Post, FetchError> {
|
||||
])
|
||||
.send()
|
||||
.await?;
|
||||
// 404/410 = gone (permanent); 429/5xx = transient and retried by fetch.
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
return match status.as_u16() {
|
||||
404 | 410 => Err(FetchError::NotFound),
|
||||
_ => Err(FetchError::Transient(format!("bsky status {status}"))),
|
||||
};
|
||||
}
|
||||
let text = response.text().await?;
|
||||
Ok(Post::from_json(&text, rkey.to_string())?)
|
||||
Post::from_json(&text, rkey.to_string())
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
@@ -61,8 +264,8 @@ impl Post {
|
||||
pub fn caption(&self) -> String {
|
||||
format!(
|
||||
"{url}\n<a href=\"{author_url}\">{author}</a>: {text}",
|
||||
url = self.url(),
|
||||
author_url = self.author_url(),
|
||||
url = encode_double_quoted_attribute(&self.url()),
|
||||
author_url = encode_double_quoted_attribute(&self.author_url()),
|
||||
author = encode_text(&self.author),
|
||||
text = encode_text(&self.text),
|
||||
)
|
||||
@@ -127,15 +330,19 @@ impl From<Post> for Fetched {
|
||||
url: url.clone(),
|
||||
author: encode_text(&post.author).into_owned(),
|
||||
author_url: author_url.clone(),
|
||||
title: encode_text(&post.text).into_owned(),
|
||||
// A post has no title: its text is all content.
|
||||
title: String::new(),
|
||||
content: encode_text(&post.text).into_owned(),
|
||||
tags: String::new(),
|
||||
});
|
||||
Fetched {
|
||||
source_url: url,
|
||||
caption: post.caption(),
|
||||
title: post.text.clone(),
|
||||
title: String::new(),
|
||||
content: post.text.clone(),
|
||||
media: post.media,
|
||||
sensitive: post.sensitive,
|
||||
site_id: "bsky",
|
||||
render_data,
|
||||
_keep_alive: None,
|
||||
}
|
||||
@@ -206,7 +413,8 @@ mod tests {
|
||||
fetched.source_url,
|
||||
"https://bsky.app/profile/user.bsky.social/post/3xxxx"
|
||||
);
|
||||
assert_eq!(fetched.title, "hello <world>");
|
||||
assert_eq!(fetched.title, "");
|
||||
assert_eq!(fetched.content, "hello <world>");
|
||||
assert_eq!(fetched.media.len(), 1);
|
||||
assert!(!fetched.sensitive);
|
||||
// display_name absent -> empty fallback
|
||||
@@ -253,6 +461,7 @@ mod tests {
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "live network: requires outbound HTTPS to public.api.bsky.app"]
|
||||
async fn live_fetch_with_photos() {
|
||||
let fetched =
|
||||
fetch_from_url("https://bsky.app/profile/asagi0398.bsky.social/post/3mqkhrq5w6k2m")
|
||||
@@ -266,6 +475,7 @@ mod tests {
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "live network: requires outbound HTTPS to public.api.bsky.app"]
|
||||
async fn live_fetch_smoke() {
|
||||
let fetched =
|
||||
fetch_from_url("https://bsky.app/profile/fu-futa.bsky.social/post/3laoveufjv224")
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
mod interface;
|
||||
mod model;
|
||||
|
||||
pub use interface::{PATTERN, Post, enabled, fetch_from_url};
|
||||
pub use interface::{
|
||||
BskySite, PATTERN, Post, cache_key, enabled, fetch_from_url, is_retryable, media_headers,
|
||||
};
|
||||
|
||||
@@ -0,0 +1,390 @@
|
||||
//! Site adapter for misskey.io notes: URL pattern, API fetch and
|
||||
//! normalization into [`Fetched`] (see [`crate::site::Site`]).
|
||||
|
||||
use super::model;
|
||||
use crate::media::Media;
|
||||
use crate::site::{FetchError, Fetched, RenderData, Site, SiteFuture};
|
||||
use html_escape::{encode_double_quoted_attribute, encode_text};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
const API_URL: &str = "https://misskey.io/api/notes/show";
|
||||
|
||||
/// Registry entry for the misskey.io adapter (see [`crate::site::Site`]).
|
||||
pub struct MisskeySite;
|
||||
|
||||
impl Site for MisskeySite {
|
||||
fn id(&self) -> &'static str {
|
||||
"misskey"
|
||||
}
|
||||
|
||||
fn pattern(&self) -> &'static Regex {
|
||||
&PATTERN
|
||||
}
|
||||
|
||||
fn cache_key(&self, url: &str) -> Option<String> {
|
||||
cache_key(url)
|
||||
}
|
||||
|
||||
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched> {
|
||||
Box::pin(async move { fetch_from_url(url).await })
|
||||
}
|
||||
}
|
||||
|
||||
pub static PATTERN: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"^(?:https?://)?misskey\.io/notes/([\w.\-~]+)").unwrap());
|
||||
|
||||
pub fn enabled() -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> {
|
||||
let caps = PATTERN.captures(url).ok_or(FetchError::NotFound)?;
|
||||
let note_id = caps.get(1).ok_or(FetchError::NotFound)?.as_str();
|
||||
let note = fetch(note_id).await?;
|
||||
Ok(note.into())
|
||||
}
|
||||
|
||||
/// Cache key for a misskey URL: `"misskey:<note id>"`. The prefix is the
|
||||
/// site id used for caption-format lookup and link-cache keys.
|
||||
pub fn cache_key(url: &str) -> Option<String> {
|
||||
PATTERN
|
||||
.captures(url)
|
||||
.map(|caps| format!("misskey:{}", &caps[1]))
|
||||
}
|
||||
|
||||
/// Misskey's fetch-retry policy: transient classes only. Not-found, blocked
|
||||
/// and parse failures are permanent.
|
||||
pub fn is_retryable(err: &FetchError) -> bool {
|
||||
matches!(err, FetchError::Http(_) | FetchError::Transient(_))
|
||||
}
|
||||
|
||||
/// misskey.io media hosts need no extra headers (verified: direct GET works).
|
||||
pub fn media_headers(_url: &str) -> Option<Vec<(&'static str, String)>> {
|
||||
None
|
||||
}
|
||||
|
||||
/// Fetches a note from misskey.io by id. The API answers client failures
|
||||
/// with HTTP 400 + `{"error":{"code":...}}` (NO_SUCH_NOTE → NotFound);
|
||||
/// everything else non-success is transient and retried by [`crate::site::fetch`].
|
||||
pub async fn fetch(note_id: &str) -> Result<model::Note, FetchError> {
|
||||
let response = crate::site::CLIENT
|
||||
.post(API_URL)
|
||||
.json(&serde_json::json!({ "noteId": note_id }))
|
||||
.send()
|
||||
.await?;
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
return Err(match status.as_u16() {
|
||||
400 => not_found_or_invalid(response).await,
|
||||
_ => FetchError::Transient(format!("misskey status {status}")),
|
||||
});
|
||||
}
|
||||
response.json().await.map_err(|e| FetchError::Site {
|
||||
site: "misskey",
|
||||
error: Box::new(e),
|
||||
})
|
||||
}
|
||||
|
||||
/// Maps a 400 response: NO_SUCH_NOTE is permanent NotFound, any other 400 is
|
||||
/// a site error (permanent — retrying a rejected request cannot succeed).
|
||||
async fn not_found_or_invalid(response: reqwest::Response) -> FetchError {
|
||||
match response.json::<serde_json::Value>().await {
|
||||
Ok(v) if v["error"]["code"] == "NO_SUCH_NOTE" => FetchError::NotFound,
|
||||
_ => FetchError::Site {
|
||||
site: "misskey",
|
||||
error: "note rejected (invalid param or private note)".into(),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// The note whose content matters: a renote shell has no text/files of its
|
||||
/// own — the embedded renote carries them.
|
||||
fn effective(note: &model::Note) -> &model::Note {
|
||||
match ¬e.renote {
|
||||
Some(renote) if note.files.is_empty() => renote,
|
||||
_ => note,
|
||||
}
|
||||
}
|
||||
|
||||
impl From<model::Note> for Fetched {
|
||||
fn from(note: model::Note) -> Self {
|
||||
let note = ¬e;
|
||||
let content = effective(note);
|
||||
let url = format!("https://misskey.io/notes/{}", note.id);
|
||||
let author = content
|
||||
.user
|
||||
.name
|
||||
.as_deref()
|
||||
.filter(|n| !n.is_empty())
|
||||
.unwrap_or(&content.user.username)
|
||||
.to_string();
|
||||
let author_url = format!("https://misskey.io/@{}", content.user.username);
|
||||
let cw = content.cw.as_deref().unwrap_or_default();
|
||||
// Notes carry hashtags inline in the text (no structured tags array);
|
||||
// a CW note gets the marker prefixed so recipients see the spoiler.
|
||||
let mut text = cw.to_string();
|
||||
if !cw.is_empty() && !text.ends_with(' ') {
|
||||
text.push(' ');
|
||||
}
|
||||
text.push_str(content.text.as_deref().unwrap_or_default().trim());
|
||||
let text = text.trim().to_string();
|
||||
|
||||
let caption = caption(&url, &author_url, &author, &text);
|
||||
let sensitive = content.cw.is_some() || content.files.iter().any(|f| f.is_sensitive);
|
||||
let media: Vec<Media> = content.files.iter().filter_map(media_from_file).collect();
|
||||
|
||||
Fetched {
|
||||
source_url: url.clone(),
|
||||
caption,
|
||||
// A note has no title: its text (CW marker included) is content.
|
||||
title: String::new(),
|
||||
content: text.clone(),
|
||||
media,
|
||||
sensitive,
|
||||
site_id: "misskey",
|
||||
render_data: Some(RenderData {
|
||||
url,
|
||||
author: encode_text(&author).into_owned(),
|
||||
author_url: author_url.clone(),
|
||||
title: String::new(),
|
||||
content: encode_text(&text).into_owned(),
|
||||
tags: String::new(),
|
||||
}),
|
||||
_keep_alive: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn caption(url: &str, author_url: &str, author: &str, text: &str) -> String {
|
||||
let url = encode_double_quoted_attribute(url);
|
||||
let author_url = encode_double_quoted_attribute(author_url);
|
||||
let author = encode_text(author);
|
||||
if text.is_empty() {
|
||||
return format!("{url}\n<a href=\"{author_url}\">{author}</a>");
|
||||
}
|
||||
format!(
|
||||
"{url}\n<a href=\"{author_url}\">{author}</a>: {text}",
|
||||
text = encode_text(text),
|
||||
)
|
||||
}
|
||||
|
||||
/// Maps a Misskey DriveFile to a [`Media`] item; unknown/audio/other types
|
||||
/// are skipped (twitter's `_ => {}` precedent). GIF must be matched before
|
||||
/// the generic image arm.
|
||||
fn media_from_file(file: &model::DriveFile) -> Option<Media> {
|
||||
let title = file.name.clone();
|
||||
match file.mime_type.as_str() {
|
||||
"image/gif" => Some(Media::Animated {
|
||||
title,
|
||||
url: file.url.clone(),
|
||||
thumbnail_url: file.thumbnail_url.clone().unwrap_or_default(),
|
||||
}),
|
||||
mime if mime.starts_with("image/") => Some(Media::Illustration {
|
||||
title,
|
||||
url: file.url.clone(),
|
||||
thumbnail_url: file.thumbnail_url.clone(),
|
||||
fallback_url: None,
|
||||
}),
|
||||
mime if mime.starts_with("video/") => Some(Media::Video {
|
||||
title,
|
||||
url: file.url.clone(),
|
||||
thumbnail_url: file.thumbnail_url.clone().unwrap_or_default(),
|
||||
}),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn note_json(json: serde_json::Value) -> model::Note {
|
||||
serde_json::from_value(json).unwrap()
|
||||
}
|
||||
|
||||
fn base_note() -> serde_json::Value {
|
||||
serde_json::json!({
|
||||
"id": "aotihl10lqrs015s",
|
||||
"text": "hello",
|
||||
"user": { "name": "ミロン", "username": "donyan47897", "host": null },
|
||||
"files": []
|
||||
})
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pattern_matches_misskey_note_urls() {
|
||||
for url in [
|
||||
"https://misskey.io/notes/aotihl10lqrs015s",
|
||||
"http://misskey.io/notes/aotihl10lqrs015s",
|
||||
"misskey.io/notes/aotihl10lqrs015s",
|
||||
] {
|
||||
assert!(PATTERN.is_match(url), "{url}");
|
||||
}
|
||||
for url in [
|
||||
"https://misskey.io/",
|
||||
"https://misskey.io/@user",
|
||||
"https://misskey.io/notes/",
|
||||
"https://x.com/user/status/123",
|
||||
] {
|
||||
assert!(!PATTERN.is_match(url), "{url}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cache_key_normalizes_variants() {
|
||||
assert_eq!(
|
||||
cache_key("https://misskey.io/notes/aotihl10lqrs015s"),
|
||||
Some("misskey:aotihl10lqrs015s".to_string())
|
||||
);
|
||||
assert_eq!(x_media_site_id("misskey:abc"), "misskey");
|
||||
}
|
||||
|
||||
fn x_media_site_id(key: &str) -> &'static str {
|
||||
crate::site::site_id_from_key(key)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn from_json_image_file() {
|
||||
let mut note = base_note();
|
||||
note["files"] = serde_json::json!([{
|
||||
"type": "image/webp",
|
||||
"url": "https://media.misskeyusercontent.jp/io/a.webp",
|
||||
"thumbnailUrl": "https://media.misskeyusercontent.jp/io/t.webp",
|
||||
"isSensitive": true,
|
||||
"name": "pic.webp"
|
||||
}]);
|
||||
let fetched: Fetched = note_json(note).into();
|
||||
assert_eq!(
|
||||
fetched.source_url,
|
||||
"https://misskey.io/notes/aotihl10lqrs015s"
|
||||
);
|
||||
assert_eq!(fetched.site_id, "misskey");
|
||||
assert_eq!(fetched.title, "");
|
||||
assert_eq!(fetched.content, "hello");
|
||||
assert!(fetched.sensitive);
|
||||
assert_eq!(fetched.media.len(), 1);
|
||||
match &fetched.media[0] {
|
||||
Media::Illustration {
|
||||
title,
|
||||
url,
|
||||
thumbnail_url,
|
||||
fallback_url,
|
||||
} => {
|
||||
assert_eq!(title.as_deref(), Some("pic.webp"));
|
||||
assert_eq!(url, "https://media.misskeyusercontent.jp/io/a.webp");
|
||||
assert_eq!(
|
||||
thumbnail_url.as_deref(),
|
||||
Some("https://media.misskeyusercontent.jp/io/t.webp")
|
||||
);
|
||||
assert!(fallback_url.is_none());
|
||||
}
|
||||
other => panic!("expected illustration, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn from_json_gif_video_and_skip_audio() {
|
||||
let mut note = base_note();
|
||||
note["files"] = serde_json::json!([
|
||||
{ "type": "audio/mpeg", "url": "https://m/a.mp3", "isSensitive": false },
|
||||
{ "type": "image/gif", "url": "https://m/a.gif", "isSensitive": false },
|
||||
{ "type": "video/webm", "url": "https://m/a.webm", "isSensitive": false }
|
||||
]);
|
||||
let fetched: Fetched = note_json(note).into();
|
||||
assert_eq!(fetched.media.len(), 2);
|
||||
assert!(
|
||||
matches!(&fetched.media[0], Media::Animated { url, .. } if url == "https://m/a.gif")
|
||||
);
|
||||
assert!(matches!(&fetched.media[1], Media::Video { url, .. } if url == "https://m/a.webm"));
|
||||
// No thumbnailUrl → empty string, not a broken URL.
|
||||
match &fetched.media[1] {
|
||||
Media::Video { thumbnail_url, .. } => assert_eq!(thumbnail_url, ""),
|
||||
other => panic!("expected video, got {other:?}"),
|
||||
}
|
||||
assert!(!fetched.sensitive);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn from_json_cw_marks_sensitive_and_prefixes_title() {
|
||||
let mut note = base_note();
|
||||
note["cw"] = serde_json::json!("spoiler");
|
||||
note["text"] = serde_json::json!("body");
|
||||
let fetched: Fetched = note_json(note).into();
|
||||
assert!(fetched.sensitive);
|
||||
assert_eq!(fetched.title, "");
|
||||
assert_eq!(fetched.content, "spoiler body");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn from_json_author_falls_back_to_username() {
|
||||
let mut note = base_note();
|
||||
note["user"] = serde_json::json!({ "name": null, "username": "donyan47897", "host": null });
|
||||
let fetched: Fetched = note_json(note).into();
|
||||
assert!(
|
||||
fetched.caption.contains("donyan47897"),
|
||||
"{}",
|
||||
fetched.caption
|
||||
);
|
||||
assert!(fetched.caption.contains("https://misskey.io/@donyan47897"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn from_json_renote_uses_embedded_content() {
|
||||
let note = serde_json::json!({
|
||||
"id": "shell0000000000",
|
||||
"text": null,
|
||||
"user": { "name": "shell", "username": "shelluser", "host": null },
|
||||
"files": [],
|
||||
"renote": {
|
||||
"id": "inner000000000",
|
||||
"text": "inner text",
|
||||
"user": { "name": "inner", "username": "inneruser", "host": null },
|
||||
"files": [
|
||||
{ "type": "image/png", "url": "https://m/i.png", "isSensitive": false }
|
||||
]
|
||||
}
|
||||
});
|
||||
let fetched: Fetched = note_json(note).into();
|
||||
assert_eq!(fetched.title, "");
|
||||
assert_eq!(fetched.content, "inner text");
|
||||
assert_eq!(fetched.media.len(), 1);
|
||||
// The source URL still points at the renote shell the user posted.
|
||||
assert_eq!(
|
||||
fetched.source_url,
|
||||
"https://misskey.io/notes/shell0000000000"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn caption_layout_matches_bsky() {
|
||||
let fetched: Fetched = note_json(base_note()).into();
|
||||
assert_eq!(
|
||||
fetched.caption,
|
||||
"https://misskey.io/notes/aotihl10lqrs015s\n<a href=\"https://misskey.io/@donyan47897\">ミロン</a>: hello"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn caption_without_text_has_no_dangling_colon() {
|
||||
let mut note = base_note();
|
||||
note["text"] = serde_json::json!(null);
|
||||
let fetched: Fetched = note_json(note).into();
|
||||
assert_eq!(
|
||||
fetched.caption,
|
||||
"https://misskey.io/notes/aotihl10lqrs015s\n<a href=\"https://misskey.io/@donyan47897\">ミロン</a>"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "live network: requires outbound HTTPS to misskey.io"]
|
||||
async fn live_fetch_reference_note() {
|
||||
let fetched = fetch_from_url("https://misskey.io/notes/aotihl10lqrs015s")
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(fetched.site_id, "misskey");
|
||||
assert_eq!(fetched.media.len(), 1);
|
||||
assert!(fetched.sensitive);
|
||||
assert!(!fetched.caption.is_empty());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
mod interface;
|
||||
mod model;
|
||||
|
||||
pub use interface::{
|
||||
MisskeySite, PATTERN, cache_key, enabled, fetch_from_url, is_retryable, media_headers,
|
||||
};
|
||||
@@ -0,0 +1,35 @@
|
||||
use serde::Deserialize;
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct Note {
|
||||
pub(crate) id: String,
|
||||
pub(crate) text: Option<String>,
|
||||
#[serde(default)]
|
||||
pub(crate) cw: Option<String>,
|
||||
pub(crate) user: User,
|
||||
#[serde(default)]
|
||||
pub(crate) files: Vec<DriveFile>,
|
||||
/// Embedded original note when this note is a renote; the shell's own
|
||||
/// text/files are usually empty and the content lives here.
|
||||
#[serde(default)]
|
||||
pub(crate) renote: Option<Box<Note>>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct User {
|
||||
pub(crate) name: Option<String>,
|
||||
pub(crate) username: String,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug)]
|
||||
pub(crate) struct DriveFile {
|
||||
#[serde(rename = "type")]
|
||||
pub(crate) mime_type: String,
|
||||
pub(crate) url: String,
|
||||
#[serde(default, rename = "thumbnailUrl")]
|
||||
pub(crate) thumbnail_url: Option<String>,
|
||||
#[serde(default, rename = "isSensitive")]
|
||||
pub(crate) is_sensitive: bool,
|
||||
#[serde(default)]
|
||||
pub(crate) name: Option<String>,
|
||||
}
|
||||
+511
-138
@@ -1,34 +1,54 @@
|
||||
//! Site fetching dispatcher and unified result types.
|
||||
//!
|
||||
//! Dispatch order: twitter → bsky → pixiv. Each site module exports a
|
||||
//! `PATTERN`, `enabled()` and `fetch_from_url()`; a future site plugs in by
|
||||
//! adding one guarded entry in [`fetch_once`].
|
||||
//! Dispatch order: twitter → bsky → misskey → pixiv → bilibili. Each site
|
||||
//! module exports a `PATTERN`, `enabled()` and `fetch_from_url()`; a future
|
||||
//! site plugs in by adding one guarded entry in `SITES`.
|
||||
|
||||
use std::fmt;
|
||||
use std::future::Future;
|
||||
use std::pin::Pin;
|
||||
use std::sync::LazyLock;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::time::Duration;
|
||||
|
||||
use regex::Regex;
|
||||
use thiserror::Error;
|
||||
|
||||
pub mod bilibili;
|
||||
pub mod bsky;
|
||||
pub mod misskey;
|
||||
pub mod pixiv;
|
||||
pub mod twitter;
|
||||
|
||||
pub use pixiv::PixivError;
|
||||
|
||||
/// The result of fetching a post: canonical URL, HTML caption, raw text,
|
||||
/// media list and spoiler flag. Produced by [`fetch`].
|
||||
/// The result of fetching a post: canonical URL, HTML caption, the post's
|
||||
/// title and body, media list and spoiler flag. Produced by [`fetch`].
|
||||
#[derive(Debug)]
|
||||
pub struct Fetched {
|
||||
/// Canonical URL: x.com/{author}/status/{id} |
|
||||
/// https://www.pixiv.net/artworks/{id} |
|
||||
/// https://bsky.app/profile/{handle}/post/{rkey}
|
||||
/// Canonical URL: `x.com/{author}/status/{id}` |
|
||||
/// `https://www.pixiv.net/artworks/{id}` |
|
||||
/// `https://bsky.app/profile/{handle}/post/{rkey}` |
|
||||
/// `https://www.bilibili.com/opus/{id}`
|
||||
pub source_url: String,
|
||||
/// The exact HTML produced by the site's caption().
|
||||
pub caption: String,
|
||||
/// Raw post text (tweet text / bsky text / pixiv title).
|
||||
/// The post's own title, where the platform has one: a pixiv artwork's
|
||||
/// title, the headline of a bilibili opus post or the title of the video
|
||||
/// an AV dynamic attaches. Empty on the platforms whose posts are text
|
||||
/// only (x/twitter, bsky, misskey) and on bilibili posts without a
|
||||
/// headline.
|
||||
pub title: String,
|
||||
/// The post's body text, as the platform exposes it: a tweet, a bsky or
|
||||
/// misskey post, a bilibili dynamic's text, a pixiv artwork's description
|
||||
/// (HTML flattened). Empty when the post has no text at all.
|
||||
pub content: String,
|
||||
pub media: Vec<crate::media::Media>,
|
||||
/// Spoiler flag for all media of this post.
|
||||
pub sensitive: bool,
|
||||
/// Site id (`"twitter"` / `"bsky"` / `"pixiv"` / `"bilibili"`): the single source of
|
||||
/// truth for site identity — caption-format lookup, cache-key prefix and
|
||||
/// the SetFormat whitelist all derive from it. Set by the producing site.
|
||||
pub site_id: &'static str,
|
||||
/// Raw values (pre-escaped) for user-customizable caption formats.
|
||||
pub(crate) render_data: Option<RenderData>,
|
||||
/// Keeps temp files (e.g. an encoded ugoira MP4) alive until the caller
|
||||
@@ -36,36 +56,52 @@ pub struct Fetched {
|
||||
pub(crate) _keep_alive: Option<tempfile::TempDir>,
|
||||
}
|
||||
|
||||
/// Pre-escaped values for `{url} {author} {author_url} {title} {tags}`
|
||||
/// placeholders in user-supplied caption formats.
|
||||
/// Values for the `{url} {author} {author_url} {title} {content} {tags}`
|
||||
/// placeholders in user-supplied caption formats, substituted by
|
||||
/// [`caption_from_fields`] as HTML text (never as an attribute value).
|
||||
///
|
||||
/// `author`, `title`, `content` and `tags` come from the site API (post
|
||||
/// text, display names, descriptions) and are HTML-escaped at construction.
|
||||
/// `url` and `author_url` stay raw: they are canonical URLs the adapter
|
||||
/// builds from numeric ids and API-constrained handles/DIDs, so they carry
|
||||
/// no escapable character — the bot's `/test` report relies on that when it
|
||||
/// embeds them.
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct RenderData {
|
||||
pub url: String,
|
||||
pub author: String,
|
||||
pub author_url: String,
|
||||
pub title: String,
|
||||
pub content: String,
|
||||
pub tags: String,
|
||||
}
|
||||
|
||||
/// The post's text as one string: title and content joined by a line break,
|
||||
/// each only when it is non-empty. This is what the sites' built-in captions
|
||||
/// show after the author line, and what the bot quotes when it is long.
|
||||
pub fn compose_text(title: &str, content: &str) -> String {
|
||||
match (title.is_empty(), content.is_empty()) {
|
||||
(false, false) => format!("{title}\n{content}"),
|
||||
(false, true) => title.to_string(),
|
||||
(true, false) => content.to_string(),
|
||||
(true, true) => String::new(),
|
||||
}
|
||||
}
|
||||
|
||||
impl Fetched {
|
||||
/// The site this post came from (used for per-site format overrides).
|
||||
/// A thin alias over [`Fetched::site_id`] kept for callers that read the
|
||||
/// site off a fetched post.
|
||||
pub fn site_name(&self) -> &'static str {
|
||||
if self.source_url.contains("x.com") || self.source_url.contains("twitter.com") {
|
||||
"twitter"
|
||||
} else if self.source_url.contains("bsky.app") {
|
||||
"bsky"
|
||||
} else if self.source_url.contains("pixiv.net") {
|
||||
"pixiv"
|
||||
} else {
|
||||
"unknown"
|
||||
}
|
||||
self.site_id
|
||||
}
|
||||
|
||||
/// Renders a user-supplied caption format. The format string is
|
||||
/// HTML-escaped in full, then the (already-escaped) placeholder values
|
||||
/// are substituted — users can structure text but never inject raw HTML
|
||||
/// or attributes. An empty/unknown format falls back to the built-in
|
||||
/// caption.
|
||||
/// caption. The result is truncated to [`MAX_CAPTION_CHARS`] (Telegram's
|
||||
/// caption limit for HTML parse mode).
|
||||
pub fn caption_with(&self, format: &str) -> String {
|
||||
match (&self.render_data, format.is_empty()) {
|
||||
(Some(data), false) => caption_from_fields(
|
||||
@@ -75,30 +111,73 @@ impl Fetched {
|
||||
&data.author,
|
||||
&data.author_url,
|
||||
&data.title,
|
||||
&data.content,
|
||||
&data.tags,
|
||||
),
|
||||
_ => self.caption.clone(),
|
||||
_ => truncate_caption(&self.caption),
|
||||
}
|
||||
}
|
||||
|
||||
/// The pre-escaped placeholder values (author, author_url, title, tags)
|
||||
/// a caller needs to rebuild a caption later, e.g. for a cached post
|
||||
/// where the [`Fetched`] is no longer available.
|
||||
pub fn render_fields(&self) -> Option<(&str, &str, &str, &str)> {
|
||||
/// The pre-escaped placeholder values (author, author_url, title,
|
||||
/// content, tags) a caller needs to rebuild a caption later, e.g. for a
|
||||
/// cached post where the [`Fetched`] is no longer available.
|
||||
pub fn render_fields(&self) -> Option<(&str, &str, &str, &str, &str)> {
|
||||
self.render_data.as_ref().map(|d| {
|
||||
(
|
||||
d.author.as_str(),
|
||||
d.author_url.as_str(),
|
||||
d.title.as_str(),
|
||||
d.content.as_str(),
|
||||
d.tags.as_str(),
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
/// Hands over the temp dir keeping locally produced media (ugoira MP4,
|
||||
/// bsky remux MP4) alive. The bot keeps it while its task may still be
|
||||
/// retried by the queue, which runs after this [`Fetched`] is dropped and
|
||||
/// its temp files would otherwise be gone. `None` when no such dir exists.
|
||||
pub fn take_keep_alive(&mut self) -> Option<tempfile::TempDir> {
|
||||
self._keep_alive.take()
|
||||
}
|
||||
}
|
||||
|
||||
/// Telegram's caption length limit (chars) for HTML parse mode; longer
|
||||
/// captions are rejected with a 400.
|
||||
pub const MAX_CAPTION_CHARS: usize = 1024;
|
||||
|
||||
/// Truncates a caption to at most [`MAX_CAPTION_CHARS`] chars, appending an
|
||||
/// ellipsis when cut. Backs off to before an unclosed HTML entity (`&`
|
||||
/// without its `;` would be malformed HTML and rejected by Telegram).
|
||||
pub fn truncate_caption(caption: &str) -> String {
|
||||
if caption.chars().count() <= MAX_CAPTION_CHARS {
|
||||
return caption.to_string();
|
||||
}
|
||||
// Leave one char for the ellipsis; floor_char_boundary lands on a char
|
||||
// edge (byte index ≤ MAX-1, so chars ≤ MAX-1).
|
||||
let mut end = caption.floor_char_boundary(MAX_CAPTION_CHARS - 1);
|
||||
// Don't split an entity: if the last '&' before `end` has no closing ';'
|
||||
// inside the kept part, cut before it.
|
||||
if let Some(amp) = caption[..end].rfind('&')
|
||||
&& !caption[amp..end].contains(';')
|
||||
{
|
||||
end = amp;
|
||||
}
|
||||
let mut s = caption[..end].to_string();
|
||||
s.push('…');
|
||||
s
|
||||
}
|
||||
|
||||
/// Renders a user-supplied caption format from raw (already-escaped) field
|
||||
/// values with the same escaping/substitution rules as
|
||||
/// [`Fetched::caption_with`]. An empty format returns `built_in` unchanged.
|
||||
/// The result is truncated to [`MAX_CAPTION_CHARS`] (Telegram's caption
|
||||
/// limit for HTML parse mode).
|
||||
///
|
||||
/// One flat argument per placeholder keeps the two callers (the fresh and the
|
||||
/// cached caption path) mirroring each other; the same shape as the bot's
|
||||
/// `debug_report`.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn caption_from_fields(
|
||||
format: &str,
|
||||
built_in: &str,
|
||||
@@ -106,95 +185,103 @@ pub fn caption_from_fields(
|
||||
author: &str,
|
||||
author_url: &str,
|
||||
title: &str,
|
||||
content: &str,
|
||||
tags: &str,
|
||||
) -> String {
|
||||
if format.is_empty() {
|
||||
return built_in.to_string();
|
||||
return truncate_caption(built_in);
|
||||
}
|
||||
let escaped = html_escape::encode_text(format).into_owned();
|
||||
escaped
|
||||
.replace("{url}", url)
|
||||
.replace("{author}", author)
|
||||
.replace("{author_url}", author_url)
|
||||
.replace("{title}", title)
|
||||
.replace("{tags}", tags)
|
||||
truncate_caption(
|
||||
&escaped
|
||||
.replace("{url}", url)
|
||||
.replace("{author}", author)
|
||||
.replace("{author_url}", author_url)
|
||||
.replace("{title}", title)
|
||||
.replace("{content}", content)
|
||||
.replace("{tags}", tags),
|
||||
)
|
||||
}
|
||||
|
||||
/// Stable per-post cache key derived from any supported URL, so variant
|
||||
/// domains (x.com / twitter.com / fxtwitter.com, mobile, `/photo/N`
|
||||
/// suffixes) map to the same post. Returns `"twitter:<id>"`,
|
||||
/// `"pixiv:<id>"` or `"bsky:<handle>/<rkey>"`.
|
||||
/// suffixes) map to the same post. Delegates to each registered site's
|
||||
/// `cache_key` (in registry order).
|
||||
pub fn cache_key(url: &str) -> Option<String> {
|
||||
if let Some(caps) = twitter::PATTERN.captures(url) {
|
||||
return Some(format!("twitter:{}", &caps[1]));
|
||||
}
|
||||
if let Some(caps) = pixiv::PATTERN.captures(url) {
|
||||
return Some(format!("pixiv:{}", &caps[1]));
|
||||
}
|
||||
if let Some(caps) = bsky::PATTERN.captures(url) {
|
||||
return Some(format!("bsky:{}/{}", &caps[1], &caps[2]));
|
||||
}
|
||||
None
|
||||
SITES.iter().find_map(|site| site.cache_key(url))
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
/// The site id carried by a cache key (`"twitter:123"` → `"twitter"`).
|
||||
/// Unknown prefixes fall back to `"unknown"`. The bot uses this on the
|
||||
/// link-cache hit path, where no [`Fetched`] is available — the same value
|
||||
/// a fresh fetch would read from [`Fetched::site_id`].
|
||||
pub fn site_id_from_key(key: &str) -> &'static str {
|
||||
let prefix = key.split(':').next().unwrap_or("");
|
||||
SITES
|
||||
.iter()
|
||||
.map(|site| site.id())
|
||||
.find(|id| *id == prefix)
|
||||
.unwrap_or("unknown")
|
||||
}
|
||||
|
||||
#[derive(Debug, Error)]
|
||||
pub enum FetchError {
|
||||
Http(reqwest::Error),
|
||||
Json(serde_json::Error),
|
||||
Pixiv(PixivError),
|
||||
#[error("http error: {0}")]
|
||||
Http(#[from] reqwest::Error),
|
||||
#[error("json error: {0}")]
|
||||
Json(#[from] serde_json::Error),
|
||||
#[error("pixiv error: {0}")]
|
||||
Pixiv(#[from] PixivError),
|
||||
/// A site-specific error from a site that keeps its own error type.
|
||||
/// Permanent by default (sites that need retryable site errors convert
|
||||
/// them to [`FetchError::Http`] / [`FetchError::Transient`] before
|
||||
/// returning). Pixiv predates this and keeps the dedicated
|
||||
/// [`FetchError::Pixiv`] variant.
|
||||
#[error("{site} error: {error}")]
|
||||
Site {
|
||||
site: &'static str,
|
||||
#[source]
|
||||
error: Box<dyn std::error::Error + Send + Sync>,
|
||||
},
|
||||
#[error("not found")]
|
||||
NotFound,
|
||||
#[error("blocked")]
|
||||
Blocked,
|
||||
/// The post exists but its content is withheld (twitter NSFW /
|
||||
/// age-restricted tweets come back as an empty `{}` from syndication).
|
||||
#[error("content withheld (sensitive)")]
|
||||
Sensitive,
|
||||
}
|
||||
|
||||
impl fmt::Display for FetchError {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
match self {
|
||||
FetchError::Http(e) => write!(f, "http error: {e}"),
|
||||
FetchError::Json(e) => write!(f, "json error: {e}"),
|
||||
FetchError::Pixiv(e) => write!(f, "pixiv error: {e}"),
|
||||
FetchError::NotFound => write!(f, "not found"),
|
||||
FetchError::Blocked => write!(f, "blocked"),
|
||||
FetchError::Sensitive => write!(f, "content withheld (sensitive)"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for FetchError {
|
||||
fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
|
||||
match self {
|
||||
FetchError::Http(e) => Some(e),
|
||||
FetchError::Json(e) => Some(e),
|
||||
FetchError::Pixiv(e) => Some(e),
|
||||
FetchError::NotFound | FetchError::Blocked | FetchError::Sensitive => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<reqwest::Error> for FetchError {
|
||||
fn from(e: reqwest::Error) -> Self {
|
||||
FetchError::Http(e)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<serde_json::Error> for FetchError {
|
||||
fn from(e: serde_json::Error) -> Self {
|
||||
FetchError::Json(e)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<PixivError> for FetchError {
|
||||
fn from(e: PixivError) -> Self {
|
||||
FetchError::Pixiv(e)
|
||||
}
|
||||
/// A download exceeded the caller's size cap (see [`download_media_limited`]).
|
||||
#[error("media too large")]
|
||||
TooLarge,
|
||||
/// A transient server-side failure (429 / 5xx); [`fetch`] retries these.
|
||||
#[error("transient: {0}")]
|
||||
Transient(String),
|
||||
/// A local I/O failure while streaming a download to disk
|
||||
/// (see [`download_media_to_file`]).
|
||||
#[error("io error: {0}")]
|
||||
Io(std::io::Error),
|
||||
}
|
||||
|
||||
/// Shared HTTP client (browser User-Agent) for twitter/bsky fetches and
|
||||
/// [`download_media`].
|
||||
pub(crate) static CLIENT: LazyLock<reqwest::Client> = LazyLock::new(|| {
|
||||
let builder = reqwest::Client::builder().user_agent("Mozilla/5.0");
|
||||
let mut builder = reqwest::Client::builder()
|
||||
.user_agent("Mozilla/5.0")
|
||||
// reqwest has no total timeout by default; a stalled connection
|
||||
// would otherwise pin a fetch/handler forever.
|
||||
.timeout(Duration::from_secs(30))
|
||||
.connect_timeout(Duration::from_secs(10));
|
||||
// Route site fetches through the same proxy the Bot API uses, so a
|
||||
// network that needs TELOXIDE_PROXY (e.g. behind the GFW) does not
|
||||
// leave site fetches dead while the bot itself works.
|
||||
if let Some(proxy) = std::env::var("TELOXIDE_PROXY")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty())
|
||||
&& let Ok(p) = reqwest::Proxy::all(&proxy)
|
||||
{
|
||||
builder = builder.proxy(p);
|
||||
}
|
||||
// Each `#[tokio::test]` runs on its own runtime; the connection pool is
|
||||
// bound to the runtime that created it, so cross-runtime reuse of idle
|
||||
// connections fails with DispatchGone. In test builds every request uses
|
||||
@@ -204,76 +291,262 @@ pub(crate) static CLIENT: LazyLock<reqwest::Client> = LazyLock::new(|| {
|
||||
builder.build().expect("failed to build HTTP client")
|
||||
});
|
||||
|
||||
/// Whether a usable `ffmpeg` binary is on PATH (probed once). Shared by the
|
||||
/// pixiv ugoira encoder and the bsky HLS remuxer.
|
||||
static FFMPEG_AVAILABLE: LazyLock<bool> = LazyLock::new(|| {
|
||||
std::process::Command::new("ffmpeg")
|
||||
.arg("-version")
|
||||
.stdout(std::process::Stdio::null())
|
||||
.stderr(std::process::Stdio::null())
|
||||
.status()
|
||||
.map(|s| s.success())
|
||||
.unwrap_or(false)
|
||||
});
|
||||
|
||||
static FFMPEG_MISSING_LOGGED: AtomicBool = AtomicBool::new(false);
|
||||
|
||||
pub(crate) fn ffmpeg_available() -> bool {
|
||||
*FFMPEG_AVAILABLE
|
||||
}
|
||||
|
||||
pub(crate) fn log_once_ffmpeg_missing() {
|
||||
if !FFMPEG_MISSING_LOGGED.swap(true, Ordering::Relaxed) {
|
||||
log::warn!("ffmpeg not found; ugoira and bsky video posts stay unsupported");
|
||||
}
|
||||
}
|
||||
|
||||
/// Site adapter: one impl per supported site (twitter / bsky / misskey /
|
||||
/// pixiv / bilibili), registered in `SITES`. All site-specific knowledge — URL pattern,
|
||||
/// cache-key format, fetch, retry policy, media-host headers, startup
|
||||
/// validation — lives in the site module; the central dispatcher only
|
||||
/// iterates the registry.
|
||||
///
|
||||
/// Async methods return a boxed future (see `SiteFuture`): `async fn` /
|
||||
/// RPITIT in traits are not dyn-compatible (verified on rustc 1.95), and
|
||||
/// `+ Send` is required since URL/queue workers spawn these futures. The
|
||||
/// site structs are stateless unit structs, so the boxed futures never
|
||||
/// borrow from `self` beyond the call's scope.
|
||||
pub trait Site: Send + Sync {
|
||||
/// Stable site id (`"twitter"` / `"bsky"` / `"misskey"` / `"pixiv"` /
|
||||
/// `"bilibili"`): caption-format lookup, cache-key prefixes and the
|
||||
/// SetFormat whitelist derive from it.
|
||||
fn id(&self) -> &'static str;
|
||||
/// URL pattern; the dispatcher's first match wins (dispatch order).
|
||||
fn pattern(&self) -> &'static Regex;
|
||||
/// Whether the site is usable (env token present, not disabled).
|
||||
fn enabled(&self) -> bool {
|
||||
true
|
||||
}
|
||||
/// Normalized cache key for a URL of this site (`None` when the URL does
|
||||
/// not match this site).
|
||||
fn cache_key(&self, url: &str) -> Option<String>;
|
||||
/// Fetches and normalizes a post.
|
||||
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched>;
|
||||
/// Retry policy for fetch errors: transient classes only.
|
||||
fn is_retryable(&self, err: &FetchError) -> bool {
|
||||
matches!(err, FetchError::Http(_) | FetchError::Transient(_))
|
||||
}
|
||||
/// Extra headers for downloading this site's media (hotlink protection,
|
||||
/// e.g. pixiv's Referer for pximg.net). Matched on the media URL, not
|
||||
/// the site pattern.
|
||||
fn media_headers(&self, _url: &str) -> Option<Vec<(&'static str, String)>> {
|
||||
None
|
||||
}
|
||||
/// Startup validation (token check etc.); failures are surfaced by
|
||||
/// [`validate_all`]. The default is a no-op.
|
||||
fn validate(&self) -> SiteFuture<'static, (), String> {
|
||||
Box::pin(async { Ok(()) })
|
||||
}
|
||||
}
|
||||
|
||||
/// A boxed, `Send` future produced by a [`Site`] async method. Boxed so the
|
||||
/// trait stays dyn-compatible; `Send` because URL/queue workers `tokio::spawn`
|
||||
/// these futures.
|
||||
type SiteFuture<'a, T, E = FetchError> = Pin<Box<dyn Future<Output = Result<T, E>> + Send + 'a>>;
|
||||
|
||||
/// The one registry of supported sites, in dispatch order (twitter → bsky →
|
||||
/// misskey → pixiv → bilibili). Adding a site = new module + one
|
||||
/// `Box::new(...)` entry here; the bot crate never lists sites itself.
|
||||
static SITES: LazyLock<Vec<Box<dyn Site>>> = LazyLock::new(|| {
|
||||
vec![
|
||||
Box::new(twitter::TwitterSite),
|
||||
Box::new(bsky::BskySite),
|
||||
Box::new(misskey::MisskeySite),
|
||||
Box::new(pixiv::PixivSite),
|
||||
Box::new(bilibili::BilibiliSite),
|
||||
]
|
||||
});
|
||||
|
||||
/// The first enabled site whose pattern matches `url`, in dispatch order.
|
||||
fn find_site(url: &str) -> Option<&'static dyn Site> {
|
||||
SITES
|
||||
.iter()
|
||||
.find(|site| site.enabled() && site.pattern().is_match(url))
|
||||
.map(|site| site.as_ref())
|
||||
}
|
||||
|
||||
/// Every supported site id, in dispatch order. The bot's SetFormat whitelist
|
||||
/// derives from this list.
|
||||
pub fn site_ids() -> Vec<&'static str> {
|
||||
SITES.iter().map(|site| site.id()).collect()
|
||||
}
|
||||
|
||||
/// Runs every enabled site's startup validation and returns the failures
|
||||
/// (site id + message). The caller logs / notifies; failing sites disable
|
||||
/// themselves (pixiv disables on a bad token).
|
||||
pub async fn validate_all() -> Vec<(&'static str, String)> {
|
||||
let mut failures = Vec::new();
|
||||
for site in SITES.iter() {
|
||||
if !site.enabled() {
|
||||
continue;
|
||||
}
|
||||
if let Err(e) = site.validate().await {
|
||||
failures.push((site.id(), e));
|
||||
}
|
||||
}
|
||||
failures
|
||||
}
|
||||
|
||||
/// Fetches a post from its URL. Returns `Ok(None)` when no site pattern
|
||||
/// matches (unsupported links are silently ignored by the bot).
|
||||
///
|
||||
/// Transient network failures are retried: 3 total attempts with 1s then 2s
|
||||
/// delays. Non-Http errors (Json/NotFound/Blocked/Pixiv) are not retried.
|
||||
/// Transient failures are retried: 3 total attempts with 1s then 2s delays.
|
||||
/// What counts as transient is the matched site's own policy (`is_retryable`
|
||||
/// — e.g. pixiv retries only network errors and 429/5xx). Permanent classes
|
||||
/// (not-found, blocked, sensitive, parse failures, pixiv 4xx/auth errors)
|
||||
/// are returned immediately; retrying them only wastes attempts against the
|
||||
/// source site.
|
||||
pub async fn fetch(url: &str) -> Result<Option<Fetched>, FetchError> {
|
||||
let mut last_http_error = None;
|
||||
for attempt in 0..3u32 {
|
||||
match fetch_once(url).await {
|
||||
Ok(Some(fetched)) => {
|
||||
log::info!(
|
||||
"fetched {url}: site {} returned {} media",
|
||||
fetch_with_attempts(url, MAX_FETCH_ATTEMPTS).await
|
||||
}
|
||||
|
||||
/// [`fetch`] without the retry backoff (one attempt). For callers with a
|
||||
/// short deadline: an inline query's answer window is measured in seconds, so
|
||||
/// the 1s + 2s retry sleeps would outlast the query the answer belongs to.
|
||||
pub async fn fetch_once(url: &str) -> Result<Option<Fetched>, FetchError> {
|
||||
fetch_with_attempts(url, 1).await
|
||||
}
|
||||
|
||||
/// Total attempts of the retried [`fetch`] (3: the initial try plus two).
|
||||
const MAX_FETCH_ATTEMPTS: u32 = 3;
|
||||
|
||||
async fn fetch_with_attempts(url: &str, attempts: u32) -> Result<Option<Fetched>, FetchError> {
|
||||
let Some(site) = find_site(url) else {
|
||||
return Ok(None);
|
||||
};
|
||||
for attempt in 0..attempts.max(1) {
|
||||
match site.fetch_from_url(url).await {
|
||||
Ok(fetched) => {
|
||||
// Per-request detail: debug only, keyed by the post id.
|
||||
log::debug!(
|
||||
"fetched [key={}]: site {} returned {} media",
|
||||
cache_key(url).unwrap_or_else(|| "?".into()),
|
||||
fetched.site_name(),
|
||||
fetched.media.len()
|
||||
);
|
||||
return Ok(Some(fetched));
|
||||
}
|
||||
Ok(None) => return Ok(None),
|
||||
Err(FetchError::Http(e)) => {
|
||||
last_http_error = Some(e);
|
||||
if attempt < 2 {
|
||||
Err(err) => {
|
||||
if site.is_retryable(&err) && attempt + 1 < attempts {
|
||||
tokio::time::sleep(Duration::from_secs(1 << attempt)).await;
|
||||
} else {
|
||||
return Err(err);
|
||||
}
|
||||
}
|
||||
Err(other) => return Err(other),
|
||||
}
|
||||
}
|
||||
Err(FetchError::Http(
|
||||
last_http_error.expect("retry loop always ran 3 attempts"),
|
||||
))
|
||||
unreachable!("retry loop always returns")
|
||||
}
|
||||
|
||||
async fn fetch_once(url: &str) -> Result<Option<Fetched>, FetchError> {
|
||||
if twitter::enabled() && twitter::PATTERN.is_match(url) {
|
||||
return Ok(Some(twitter::fetch_from_url(url).await?));
|
||||
/// Applies every site's media-header rule to a download request (pixiv's
|
||||
/// `Referer` for pximg.net hotlink protection). Sites contribute via their
|
||||
/// `media_headers(url)` — the central download code carries no per-site logic.
|
||||
fn apply_media_headers(mut request: reqwest::RequestBuilder, url: &str) -> reqwest::RequestBuilder {
|
||||
for site in SITES.iter() {
|
||||
if let Some(headers) = site.media_headers(url) {
|
||||
for (name, value) in headers {
|
||||
request = request.header(name, value);
|
||||
}
|
||||
}
|
||||
}
|
||||
if bsky::enabled() && bsky::PATTERN.is_match(url) {
|
||||
return Ok(Some(bsky::fetch_from_url(url).await?));
|
||||
}
|
||||
if pixiv::enabled() && pixiv::PATTERN.is_match(url) {
|
||||
return Ok(Some(pixiv::fetch_from_url(url).await?));
|
||||
}
|
||||
Ok(None)
|
||||
request
|
||||
}
|
||||
|
||||
/// Downloads media bytes for the bot's upload fallback: when Telegram's own
|
||||
/// fetch of a media URL is blocked (hotlink protection), the bot downloads
|
||||
/// the file itself and uploads it via multipart. Site-appropriate headers:
|
||||
/// pixiv image hosts need the `Referer` header.
|
||||
/// the file itself and uploads it via multipart. Site-appropriate headers
|
||||
/// come from each site's `media_headers` (pixiv image hosts need `Referer`).
|
||||
/// Returns the Content-Length of a media URL, or `None` when the server does
|
||||
/// not report one. Used to check whether a file fits Telegram's size limits
|
||||
/// before downloading/uploading it.
|
||||
pub async fn media_size(url: &str) -> Result<Option<u64>, FetchError> {
|
||||
let mut request = CLIENT.get(url);
|
||||
let lower = url.to_ascii_lowercase();
|
||||
if lower.contains("pximg.net") {
|
||||
request = request.header("Referer", "https://www.pixiv.net/");
|
||||
}
|
||||
let response = request.send().await?;
|
||||
let response = apply_media_headers(CLIENT.get(url), url)
|
||||
.send()
|
||||
.await?
|
||||
.error_for_status()?;
|
||||
Ok(response.content_length())
|
||||
}
|
||||
|
||||
pub async fn download_media(url: &str) -> Result<bytes::Bytes, FetchError> {
|
||||
let mut request = CLIENT.get(url);
|
||||
let lower = url.to_ascii_lowercase();
|
||||
if lower.contains("pximg.net") {
|
||||
request = request.header("Referer", "https://www.pixiv.net/");
|
||||
/// Downloads a media file with a hard size cap: the body is streamed and the
|
||||
/// download aborts with [`FetchError::TooLarge`] the moment the cap is
|
||||
/// crossed (or when a declared Content-Length already exceeds it). Keeps the
|
||||
/// bot from buffering arbitrarily large bodies into memory.
|
||||
pub async fn download_media_limited(url: &str, max_bytes: u64) -> Result<bytes::Bytes, FetchError> {
|
||||
let response = apply_media_headers(CLIENT.get(url), url)
|
||||
.send()
|
||||
.await?
|
||||
.error_for_status()?;
|
||||
if let Some(len) = response.content_length()
|
||||
&& len > max_bytes
|
||||
{
|
||||
return Err(FetchError::TooLarge);
|
||||
}
|
||||
let response = request.send().await?;
|
||||
Ok(response.bytes().await?)
|
||||
let mut response = response;
|
||||
let mut buf = Vec::new();
|
||||
while let Some(chunk) = response.chunk().await? {
|
||||
buf.extend_from_slice(&chunk);
|
||||
if buf.len() as u64 > max_bytes {
|
||||
return Err(FetchError::TooLarge);
|
||||
}
|
||||
}
|
||||
Ok(bytes::Bytes::from(buf))
|
||||
}
|
||||
|
||||
pub async fn download_media(url: &str) -> Result<bytes::Bytes, FetchError> {
|
||||
download_media_limited(url, u64::MAX).await
|
||||
}
|
||||
|
||||
/// Streams a download to `out`, aborting with [`FetchError::TooLarge`] the
|
||||
/// moment the body crosses `max_bytes` (or when a declared Content-Length
|
||||
/// already exceeds it). Unlike [`download_media_limited`] the body is never
|
||||
/// buffered in memory — used for large files (e.g. the pixiv ugoira frame
|
||||
/// zip, which can be hundreds of MB) that would otherwise spike RAM.
|
||||
/// Returns the number of bytes written.
|
||||
pub async fn download_media_to_file(
|
||||
url: &str,
|
||||
max_bytes: u64,
|
||||
out: &mut std::fs::File,
|
||||
) -> Result<u64, FetchError> {
|
||||
use std::io::Write;
|
||||
let response = apply_media_headers(CLIENT.get(url), url)
|
||||
.send()
|
||||
.await?
|
||||
.error_for_status()?;
|
||||
if let Some(len) = response.content_length()
|
||||
&& len > max_bytes
|
||||
{
|
||||
return Err(FetchError::TooLarge);
|
||||
}
|
||||
let mut response = response;
|
||||
let mut total: u64 = 0;
|
||||
while let Some(chunk) = response.chunk().await? {
|
||||
total += chunk.len() as u64;
|
||||
if total > max_bytes {
|
||||
return Err(FetchError::TooLarge);
|
||||
}
|
||||
out.write_all(&chunk).map_err(FetchError::Io)?;
|
||||
}
|
||||
Ok(total)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -302,33 +575,127 @@ mod tests {
|
||||
cache_key("https://bsky.app/profile/handle.example/post/3lorem"),
|
||||
Some("bsky:handle.example/3lorem".into())
|
||||
);
|
||||
assert_eq!(
|
||||
cache_key("https://t.bilibili.com/1245284537985925159"),
|
||||
Some("bilibili:1245284537985925159".into())
|
||||
);
|
||||
assert_eq!(cache_key("https://example.com/not-a-post"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn site_id_from_key_parses_prefix() {
|
||||
assert_eq!(site_id_from_key("twitter:123"), "twitter");
|
||||
assert_eq!(site_id_from_key("pixiv:123"), "pixiv");
|
||||
assert_eq!(site_id_from_key("bsky:handle.example/3lorem"), "bsky");
|
||||
assert_eq!(site_id_from_key("bilibili:123"), "bilibili");
|
||||
assert_eq!(site_id_from_key("unknown:1"), "unknown");
|
||||
assert_eq!(site_id_from_key("no-colon"), "unknown");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn registry_lists_all_sites_in_dispatch_order() {
|
||||
assert_eq!(
|
||||
site_ids(),
|
||||
vec!["twitter", "bsky", "misskey", "pixiv", "bilibili"]
|
||||
);
|
||||
// Enabled sites dispatch; unsupported URLs never match.
|
||||
assert!(find_site("https://x.com/u/status/1").is_some());
|
||||
assert!(find_site("https://misskey.io/notes/abc").is_some());
|
||||
assert!(find_site("https://t.bilibili.com/1245284537985925159").is_some());
|
||||
assert!(find_site("https://example.com/x").is_none());
|
||||
// Cache keys are pattern-driven, independent of the enabled() gate
|
||||
// (pixiv is disabled in tests without PIXIV_REFRESH_TOKEN).
|
||||
assert_eq!(
|
||||
cache_key("https://www.pixiv.net/artworks/1"),
|
||||
Some("pixiv:1".into())
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn site_error_variant_displays_and_sources() {
|
||||
use std::error::Error as _;
|
||||
let err = FetchError::Site {
|
||||
site: "example",
|
||||
error: Box::new(std::io::Error::other("boom")),
|
||||
};
|
||||
assert_eq!(err.to_string(), "example error: boom");
|
||||
assert!(err.source().is_some());
|
||||
// Permanent by default: no site's is_retryable matches it.
|
||||
assert!(!twitter::is_retryable(&err));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn caption_from_fields_substitutes_and_escapes() {
|
||||
// The format string is escaped, the field values are substituted
|
||||
// verbatim (callers pass the already-escaped render data).
|
||||
let out = caption_from_fields(
|
||||
"see {author} at {url} — {title}",
|
||||
"see {author} at {url} — {title}: {content}",
|
||||
"",
|
||||
"https://x.com/u/status/1",
|
||||
"A & B",
|
||||
"https://x.com/u",
|
||||
"hello <world>",
|
||||
"the body",
|
||||
"",
|
||||
);
|
||||
assert_eq!(
|
||||
out,
|
||||
"see A & B at https://x.com/u/status/1 — hello <world>"
|
||||
"see A & B at https://x.com/u/status/1 — hello <world>: the body"
|
||||
);
|
||||
// Empty format keeps the built-in caption untouched.
|
||||
assert_eq!(
|
||||
caption_from_fields("", "built-in", "u", "a", "au", "t", "g"),
|
||||
caption_from_fields("", "built-in", "u", "a", "au", "t", "c", "g"),
|
||||
"built-in"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compose_text_joins_title_and_content() {
|
||||
assert_eq!(compose_text("标题", "正文"), "标题\n正文");
|
||||
assert_eq!(compose_text("标题", ""), "标题");
|
||||
assert_eq!(compose_text("", "正文"), "正文");
|
||||
assert_eq!(compose_text("", ""), "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn truncate_caption_keeps_short_text() {
|
||||
assert_eq!(truncate_caption("short"), "short");
|
||||
// Exactly at the limit: untouched.
|
||||
let exact = "x".repeat(MAX_CAPTION_CHARS);
|
||||
assert_eq!(truncate_caption(&exact), exact);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn truncate_caption_cuts_long_text_with_ellipsis() {
|
||||
let long = "x".repeat(MAX_CAPTION_CHARS + 100);
|
||||
let out = truncate_caption(&long);
|
||||
assert!(
|
||||
out.chars().count() <= MAX_CAPTION_CHARS,
|
||||
"len {}",
|
||||
out.chars().count()
|
||||
);
|
||||
assert!(out.ends_with('…'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn truncate_caption_does_not_split_an_html_entity() {
|
||||
// An entity crossing the cut must not be left half-open (& without ;).
|
||||
let mut long = "a".repeat(MAX_CAPTION_CHARS - 4);
|
||||
long.push_str("&bbbb");
|
||||
let out = truncate_caption(&long);
|
||||
assert!(out.chars().count() <= MAX_CAPTION_CHARS);
|
||||
assert!(!out.contains("&"), "half entity left: {out:?}");
|
||||
assert!(!out.ends_with('&'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn truncate_caption_handles_multibyte_boundary() {
|
||||
// Multi-byte chars near the cut must not panic (char-boundary cut).
|
||||
let long = "界".repeat(MAX_CAPTION_CHARS + 10);
|
||||
let out = truncate_caption(&long);
|
||||
assert!(out.chars().count() <= MAX_CAPTION_CHARS);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn unsupported_url_returns_none() {
|
||||
let result = fetch("https://example.com/some/article").await;
|
||||
@@ -345,7 +712,13 @@ mod tests {
|
||||
async fn download_media_pixiv_original_with_referer() {
|
||||
// Proves the Referer header is attached for i.pximg.net: a header-less
|
||||
// GET to a pixiv original URL is rejected with 403.
|
||||
if std::env::var("PIXIV_REFRESH_TOKEN").is_err() {
|
||||
// Empty-string check too: an unset CI secret arrives as "" (GitHub
|
||||
// Actions), which would otherwise run the test tokenless and fail.
|
||||
if std::env::var("PIXIV_REFRESH_TOKEN")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty())
|
||||
.is_none()
|
||||
{
|
||||
eprintln!("skipping: no PIXIV_REFRESH_TOKEN");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -8,11 +8,11 @@ use super::model::{IllustrationModel, TypeModel, UgoiraMetadataModel};
|
||||
use crate::media::Media;
|
||||
use crate::site::FetchError;
|
||||
use std::env;
|
||||
use std::fmt;
|
||||
use std::io::{Cursor, Read};
|
||||
use std::io::Read;
|
||||
use std::sync::LazyLock;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::time::{Duration, SystemTime};
|
||||
use thiserror::Error;
|
||||
|
||||
const AUTH_TOKEN_URL: &str = "https://oauth.secure.pixiv.net/auth/token";
|
||||
const APP_API_URL: &str = "https://app-api.pixiv.net";
|
||||
@@ -23,48 +23,24 @@ const APP_USER_AGENT: &str = "PixivIOSApp/7.13.3 (iOS 14.6; iPhone13,2)";
|
||||
/// Token refresh safe margin (seconds).
|
||||
const TOKEN_REFRESH_SAFE_MARGIN: u64 = 300;
|
||||
|
||||
#[derive(Debug)]
|
||||
#[derive(Debug, Error)]
|
||||
pub enum PixivError {
|
||||
/// No refresh token available (PIXIV_REFRESH_TOKEN unset).
|
||||
#[error("pixiv: no authentication")]
|
||||
NoAuth,
|
||||
Http(reqwest::Error),
|
||||
Json(serde_json::Error),
|
||||
#[error("pixiv http error: {0}")]
|
||||
Http(#[from] reqwest::Error),
|
||||
#[error("pixiv json error: {0}")]
|
||||
Json(#[from] serde_json::Error),
|
||||
/// Non-2xx HTTP status from the app API. The code lets [`crate::site::fetch`]
|
||||
/// retry only transient classes (429 / 5xx) instead of burning attempts on
|
||||
/// permanent 4xx (bad token, forbidden, not found).
|
||||
#[error("pixiv status {0}")]
|
||||
Status(u16),
|
||||
#[error("pixiv api error: {0}")]
|
||||
Api(String),
|
||||
}
|
||||
|
||||
impl fmt::Display for PixivError {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
match self {
|
||||
PixivError::NoAuth => write!(f, "pixiv: no authentication"),
|
||||
PixivError::Http(e) => write!(f, "pixiv http error: {e}"),
|
||||
PixivError::Json(e) => write!(f, "pixiv json error: {e}"),
|
||||
PixivError::Api(message) => write!(f, "pixiv api error: {message}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for PixivError {
|
||||
fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
|
||||
match self {
|
||||
PixivError::Http(e) => Some(e),
|
||||
PixivError::Json(e) => Some(e),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<reqwest::Error> for PixivError {
|
||||
fn from(e: reqwest::Error) -> Self {
|
||||
PixivError::Http(e)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<serde_json::Error> for PixivError {
|
||||
fn from(e: serde_json::Error) -> Self {
|
||||
PixivError::Json(e)
|
||||
}
|
||||
}
|
||||
|
||||
/// Native pixiv app-API client.
|
||||
pub struct PixivAPI {
|
||||
refresh_token: String,
|
||||
@@ -136,6 +112,9 @@ impl PixivAPI {
|
||||
.bearer_auth(access_token)
|
||||
.send()
|
||||
.await?;
|
||||
if !response.status().is_success() {
|
||||
return Err(PixivError::Status(response.status().as_u16()));
|
||||
}
|
||||
let json: serde_json::Value = serde_json::from_str(&response.text().await?)?;
|
||||
if json.get("error").is_some() {
|
||||
let message = json
|
||||
@@ -186,6 +165,9 @@ impl PixivAPI {
|
||||
.bearer_auth(access_token)
|
||||
.send()
|
||||
.await?;
|
||||
if !response.status().is_success() {
|
||||
return Err(PixivError::Status(response.status().as_u16()));
|
||||
}
|
||||
let json: serde_json::Value = serde_json::from_str(&response.text().await?)?;
|
||||
if json.get("error").is_some() {
|
||||
let message = json
|
||||
@@ -207,8 +189,8 @@ impl PixivAPI {
|
||||
&self,
|
||||
illust_id: u64,
|
||||
) -> Result<Option<(String, tempfile::TempDir)>, PixivError> {
|
||||
if !ffmpeg_available() {
|
||||
log_once_ffmpeg_missing();
|
||||
if !crate::site::ffmpeg_available() {
|
||||
crate::site::log_once_ffmpeg_missing();
|
||||
return Ok(None);
|
||||
}
|
||||
let metadata = self.ugoira_metadata(illust_id).await?;
|
||||
@@ -222,7 +204,14 @@ impl PixivAPI {
|
||||
let Some(zip_url) = zip_url else {
|
||||
return Ok(None);
|
||||
};
|
||||
let zip_bytes = crate::site::download_media(&zip_url)
|
||||
// Stream the frame zip to a temp file instead of buffering it in
|
||||
// memory: ugoira zips can be hundreds of MB, and the old
|
||||
// download_media_limited path spiked RAM up to the size cap.
|
||||
let mut zip_file = tempfile::Builder::new()
|
||||
.suffix(".zip")
|
||||
.tempfile()
|
||||
.map_err(|e| PixivError::Api(format!("temp zip failed: {e}")))?;
|
||||
crate::site::download_media_to_file(&zip_url, 512 * 1024 * 1024, zip_file.as_file_mut())
|
||||
.await
|
||||
.map_err(|e| match e {
|
||||
FetchError::Http(e) => PixivError::Http(e),
|
||||
@@ -235,26 +224,54 @@ impl PixivAPI {
|
||||
let out_dir = tempfile::tempdir().map_err(|e| e.to_string())?;
|
||||
|
||||
// Extract frames to canonical zero-padded names; pixiv ugoira
|
||||
// frames are uniformly jpg or png per artwork.
|
||||
let mut archive = zip::ZipArchive::new(Cursor::new(zip_bytes))
|
||||
.map_err(|e| format!("unzip: {e}"))?;
|
||||
// pixiv ugoira frames are uniformly jpg or png per artwork; take
|
||||
// the extension from the first entry.
|
||||
let extension = if archive.len() > 0 {
|
||||
let first_name = archive
|
||||
.by_index(0)
|
||||
.map_err(|e| e.to_string())?
|
||||
.name()
|
||||
.to_string();
|
||||
first_name.rsplit('.').next().unwrap_or("jpg").to_string()
|
||||
// frames are uniformly jpg or png per artwork. The zip is read
|
||||
// from disk; `zip_file` stays alive for the whole extraction.
|
||||
let mut archive = zip::ZipArchive::new(
|
||||
std::fs::File::open(zip_file.path()).map_err(|e| e.to_string())?,
|
||||
)
|
||||
.map_err(|e| format!("unzip: {e}"))?;
|
||||
if archive.is_empty() {
|
||||
return Err("empty frame zip".to_string());
|
||||
}
|
||||
// Uniform jpg or png per artwork; sniff the first entry's
|
||||
// magic bytes instead of trusting its filename.
|
||||
let first = archive.by_index(0).map_err(|e| e.to_string())?;
|
||||
let mut first_bytes = Vec::new();
|
||||
first
|
||||
.take(64 * 1024 * 1024 + 1)
|
||||
.read_to_end(&mut first_bytes)
|
||||
.map_err(|e| e.to_string())?;
|
||||
if first_bytes.len() > 64 * 1024 * 1024 {
|
||||
return Err("frame exceeds size cap".to_string());
|
||||
}
|
||||
let extension = if first_bytes.starts_with(&[0xFF, 0xD8]) {
|
||||
"jpg"
|
||||
} else if first_bytes.starts_with(b"\x89PNG") {
|
||||
"png"
|
||||
} else {
|
||||
"jpg".to_string()
|
||||
"jpg"
|
||||
};
|
||||
let mut count = 0usize;
|
||||
for i in 0..archive.len() {
|
||||
let mut entry = archive.by_index(i).map_err(|e| e.to_string())?;
|
||||
{
|
||||
let path = frames_dir
|
||||
.path()
|
||||
.join(format!("img_{count:05}.{extension}"));
|
||||
std::fs::write(&path, &first_bytes).map_err(|e| e.to_string())?;
|
||||
count += 1;
|
||||
}
|
||||
for i in 1..archive.len() {
|
||||
let entry = archive.by_index(i).map_err(|e| e.to_string())?;
|
||||
if entry.size() > 64 * 1024 * 1024 {
|
||||
return Err(format!("frame {i} exceeds size cap"));
|
||||
}
|
||||
let mut bytes = Vec::new();
|
||||
entry.read_to_end(&mut bytes).map_err(|e| e.to_string())?;
|
||||
entry
|
||||
.take(64 * 1024 * 1024 + 1)
|
||||
.read_to_end(&mut bytes)
|
||||
.map_err(|e| e.to_string())?;
|
||||
if bytes.len() > 64 * 1024 * 1024 {
|
||||
return Err(format!("frame {i} exceeds size cap"));
|
||||
}
|
||||
let path = frames_dir
|
||||
.path()
|
||||
.join(format!("img_{count:05}.{extension}"));
|
||||
@@ -304,7 +321,10 @@ impl PixivAPI {
|
||||
Ok((output.to_string_lossy().into_owned(), out_dir))
|
||||
})
|
||||
.await
|
||||
.expect("ugoira encode worker panicked");
|
||||
.map_err(|e| {
|
||||
log::error!("ugoira encode worker panicked for {illust_id}: {e}");
|
||||
PixivError::Api(format!("ugoira worker failed: {e}"))
|
||||
})?;
|
||||
match result {
|
||||
Ok(pair) => Ok(Some(pair)),
|
||||
Err(message) => {
|
||||
@@ -315,28 +335,6 @@ impl PixivAPI {
|
||||
}
|
||||
}
|
||||
|
||||
static FFMPEG_AVAILABLE: LazyLock<bool> = LazyLock::new(|| {
|
||||
std::process::Command::new("ffmpeg")
|
||||
.arg("-version")
|
||||
.stdout(std::process::Stdio::null())
|
||||
.stderr(std::process::Stdio::null())
|
||||
.status()
|
||||
.map(|s| s.success())
|
||||
.unwrap_or(false)
|
||||
});
|
||||
|
||||
static FFMPEG_MISSING_LOGGED: AtomicBool = AtomicBool::new(false);
|
||||
|
||||
fn ffmpeg_available() -> bool {
|
||||
*FFMPEG_AVAILABLE
|
||||
}
|
||||
|
||||
fn log_once_ffmpeg_missing() {
|
||||
if !FFMPEG_MISSING_LOGGED.swap(true, Ordering::Relaxed) {
|
||||
log::warn!("ffmpeg not found; pixiv ugoira posts stay unsupported");
|
||||
}
|
||||
}
|
||||
|
||||
/// pixiv3-rs replacement: `None` when `PIXIV_REFRESH_TOKEN` is unset.
|
||||
static PIXIV_CLIENT: LazyLock<Option<PixivAPI>> =
|
||||
LazyLock::new(|| env::var("PIXIV_REFRESH_TOKEN").ok().map(PixivAPI::new));
|
||||
@@ -380,16 +378,31 @@ mod tests {
|
||||
use super::*;
|
||||
use dotenv::dotenv;
|
||||
|
||||
/// Skips when `PIXIV_REFRESH_TOKEN` is absent or empty (CI without the
|
||||
/// secret must stay green; GitHub Actions exposes an unset secret as an
|
||||
/// empty string, so `is_err()` alone is not enough).
|
||||
fn require_pixiv_token() -> bool {
|
||||
std::env::var("PIXIV_REFRESH_TOKEN")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty())
|
||||
.is_some()
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_fetch() {
|
||||
dotenv().ok();
|
||||
if !require_pixiv_token() {
|
||||
eprintln!("skipping: no PIXIV_REFRESH_TOKEN");
|
||||
return;
|
||||
}
|
||||
let result = fetch(126839080).await;
|
||||
assert!(result.is_ok());
|
||||
println!("{:#?}", result);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn validate_with_bogus_token_fails() {
|
||||
#[ignore = "live network: requires outbound HTTPS to oauth.secure.pixiv.net"]
|
||||
async fn live_validate_with_bogus_token_fails() {
|
||||
dotenv().ok();
|
||||
// A bogus token must surface as Api error (invalid_grant), not panic.
|
||||
let client = PixivAPI::new("bogus_token_for_testing".to_string());
|
||||
|
||||
@@ -1,18 +1,65 @@
|
||||
use super::model::{IllustrationModel, TypeModel};
|
||||
use crate::media::Media;
|
||||
use crate::site::{FetchError, Fetched};
|
||||
use html_escape::encode_text;
|
||||
use crate::site::{FetchError, Fetched, PixivError, Site, SiteFuture};
|
||||
use html_escape::{encode_double_quoted_attribute, encode_text};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static PATTERN: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"(?:www\.)?pixiv\.net/(?:en/)?(?:(?:i|artworks)/|member_illust\.php\?(?:mode=[a-z_]*&)?illust_id=)(\d+)").unwrap()
|
||||
Regex::new(r"^(?:https?://)?(?:www\.)?pixiv\.net/(?:en/)?(?:(?:i|artworks)/|member_illust\.php\?(?:mode=[a-z_]*&)?illust_id=)(\d+)").unwrap()
|
||||
});
|
||||
|
||||
pub fn enabled() -> bool {
|
||||
super::api::enabled()
|
||||
}
|
||||
|
||||
/// Registry entry for the pixiv adapter (see [`crate::site::Site`]).
|
||||
pub struct PixivSite;
|
||||
|
||||
impl Site for PixivSite {
|
||||
fn id(&self) -> &'static str {
|
||||
"pixiv"
|
||||
}
|
||||
|
||||
fn pattern(&self) -> &'static Regex {
|
||||
&PATTERN
|
||||
}
|
||||
|
||||
fn enabled(&self) -> bool {
|
||||
enabled()
|
||||
}
|
||||
|
||||
fn cache_key(&self, url: &str) -> Option<String> {
|
||||
cache_key(url)
|
||||
}
|
||||
|
||||
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched> {
|
||||
Box::pin(async move { fetch_from_url(url).await })
|
||||
}
|
||||
|
||||
fn is_retryable(&self, err: &FetchError) -> bool {
|
||||
is_retryable(err)
|
||||
}
|
||||
|
||||
fn media_headers(&self, url: &str) -> Option<Vec<(&'static str, String)>> {
|
||||
media_headers(url)
|
||||
}
|
||||
|
||||
fn validate(&self) -> SiteFuture<'static, (), String> {
|
||||
Box::pin(async {
|
||||
match super::api::validate().await {
|
||||
Ok(()) => Ok(()),
|
||||
Err(e) => {
|
||||
// Keep the old behavior: a failed login disables pixiv
|
||||
// for the rest of this process.
|
||||
super::api::disable();
|
||||
Err(format!("{e}"))
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> {
|
||||
let id = PATTERN
|
||||
.captures(url)
|
||||
@@ -23,10 +70,97 @@ pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> {
|
||||
Ok(super::api::fetch(id).await?.into())
|
||||
}
|
||||
|
||||
/// Cache key for a pixiv URL: `"pixiv:<id>"`. The prefix is the site id used
|
||||
/// for caption-format lookup and link-cache keys.
|
||||
pub fn cache_key(url: &str) -> Option<String> {
|
||||
PATTERN
|
||||
.captures(url)
|
||||
.map(|caps| format!("pixiv:{}", &caps[1]))
|
||||
}
|
||||
|
||||
/// Pixiv's fetch-retry policy: transient classes only — network errors and
|
||||
/// HTTP 429/5xx. Permanent 4xx (bad/expired token, forbidden, not found),
|
||||
/// API/auth errors, unparseable bodies and missing auth are not retried.
|
||||
pub fn is_retryable(err: &FetchError) -> bool {
|
||||
match err {
|
||||
FetchError::Http(_) | FetchError::Transient(_) => true,
|
||||
FetchError::Pixiv(e) => match e {
|
||||
PixivError::Http(_) => true,
|
||||
PixivError::Status(code) if *code == 429 || *code >= 500 => true,
|
||||
PixivError::Status(_)
|
||||
| PixivError::Api(_)
|
||||
| PixivError::Json(_)
|
||||
| PixivError::NoAuth => false,
|
||||
},
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// pximg.net is hotlink-protected: downloads must carry the pixiv Referer.
|
||||
/// The match is on the media host, not the site PATTERN — pixiv's PATTERN
|
||||
/// only matches `pixiv.net/artworks/...`, never `i.pximg.net`.
|
||||
pub fn media_headers(url: &str) -> Option<Vec<(&'static str, String)>> {
|
||||
if url.to_ascii_lowercase().contains("pximg.net") {
|
||||
Some(vec![("Referer", "https://www.pixiv.net/".to_string())])
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
/// Flattens the app API's HTML description into plain text: `<br>` (and `<p>`)
|
||||
/// become line breaks, other tags are dropped, entities decoded, the ends
|
||||
/// trimmed. A caption shows text, not markup, so the author's `<a href>` links
|
||||
/// contribute their link text only.
|
||||
fn flatten_html(raw: &str) -> String {
|
||||
let mut out = String::with_capacity(raw.len());
|
||||
let mut chars = raw.chars().peekable();
|
||||
while let Some(c) = chars.next() {
|
||||
// Only `<` followed by `/` or a letter opens a tag — a bare `<` in
|
||||
// prose ("2 < 3") is text.
|
||||
let opens_tag = c == '<'
|
||||
&& chars
|
||||
.peek()
|
||||
.is_some_and(|next| *next == '/' || next.is_ascii_alphabetic());
|
||||
if !opens_tag {
|
||||
out.push(c);
|
||||
continue;
|
||||
}
|
||||
let mut tag = String::new();
|
||||
let mut closed = false;
|
||||
for c in chars.by_ref() {
|
||||
if c == '>' {
|
||||
closed = true;
|
||||
break;
|
||||
}
|
||||
tag.push(c);
|
||||
}
|
||||
if !closed {
|
||||
// Unclosed `<…`: keep it as text rather than dropping the tail.
|
||||
out.push('<');
|
||||
out.push_str(&tag);
|
||||
break;
|
||||
}
|
||||
// `<br>`, `<br/>`, `<br />` with or without attributes, and both
|
||||
// halves of a paragraph break the line; everything else is dropped.
|
||||
let tag = tag
|
||||
.trim()
|
||||
.trim_start_matches('/')
|
||||
.trim_end_matches('/')
|
||||
.trim()
|
||||
.to_ascii_lowercase();
|
||||
if tag == "p" || tag.starts_with("br") {
|
||||
out.push('\n');
|
||||
}
|
||||
}
|
||||
html_escape::decode_html_entities(&out).trim().to_string()
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub struct Illustration {
|
||||
id: String,
|
||||
title: String,
|
||||
/// The artwork's description, HTML flattened to plain text.
|
||||
content: String,
|
||||
author: String,
|
||||
author_id: String,
|
||||
tags: Vec<String>,
|
||||
@@ -48,9 +182,9 @@ impl Illustration {
|
||||
pub fn caption(&self) -> String {
|
||||
format!(
|
||||
"<a href=\"{url}\">{title}</a> / <a href=\"{author_url}\">{author}</a>\n{tags}",
|
||||
url = self.url(),
|
||||
url = encode_double_quoted_attribute(&self.url()),
|
||||
title = encode_text(&self.title),
|
||||
author_url = self.author_url(),
|
||||
author_url = encode_double_quoted_attribute(&self.author_url()),
|
||||
author = encode_text(&self.author),
|
||||
tags = encode_text(
|
||||
&self
|
||||
@@ -66,6 +200,7 @@ impl Illustration {
|
||||
pub fn from_model(model: &IllustrationModel) -> Self {
|
||||
let id = model.id.to_string();
|
||||
let title = model.title.clone();
|
||||
let content = flatten_html(&model.caption);
|
||||
let author = model.user.name.clone();
|
||||
let author_id = model.user.id.to_string();
|
||||
let mut tags: Vec<String> = model.tags.iter().map(|tag| tag.name.clone()).collect();
|
||||
@@ -109,6 +244,7 @@ impl Illustration {
|
||||
Self {
|
||||
id,
|
||||
title,
|
||||
content,
|
||||
author,
|
||||
author_id,
|
||||
tags,
|
||||
@@ -134,14 +270,17 @@ impl From<Illustration> for Fetched {
|
||||
author: encode_text(&illustration.author).into_owned(),
|
||||
author_url: author_url.clone(),
|
||||
title: encode_text(&illustration.title).into_owned(),
|
||||
content: encode_text(&illustration.content).into_owned(),
|
||||
tags: encode_text(&tags).into_owned(),
|
||||
});
|
||||
Fetched {
|
||||
source_url: url,
|
||||
caption: illustration.caption(),
|
||||
title: illustration.title.clone(),
|
||||
content: illustration.content.clone(),
|
||||
media: illustration.media,
|
||||
sensitive: illustration.nsfw,
|
||||
site_id: "pixiv",
|
||||
render_data,
|
||||
_keep_alive: illustration._keep_alive,
|
||||
}
|
||||
@@ -177,6 +316,7 @@ mod tests {
|
||||
"illust": {
|
||||
"id": 123,
|
||||
"title": "Art <title>",
|
||||
"caption": "一行说明<br />二行 <a href=\"https://x.example/\">链接</a> & 结尾",
|
||||
"type": type_,
|
||||
"image_urls": {
|
||||
"medium": "medium.jpg",
|
||||
@@ -199,6 +339,44 @@ mod tests {
|
||||
Illustration::from_model(&model)
|
||||
}
|
||||
|
||||
/// The description arrives as HTML and becomes plain-text content: breaks
|
||||
/// kept, tags dropped (links keep their text), entities decoded.
|
||||
#[test]
|
||||
fn from_json_maps_description_to_content() {
|
||||
let v = illust_json("illust", 1, None, Some("o.jpg"), vec![], 0);
|
||||
let illustration = parse(v);
|
||||
assert_eq!(illustration.content, "一行说明\n二行 链接 & 结尾");
|
||||
|
||||
let fetched: Fetched = illustration.into();
|
||||
assert_eq!(fetched.title, "Art <title>");
|
||||
assert_eq!(fetched.content, "一行说明\n二行 链接 & 结尾");
|
||||
// The built-in caption keeps its layout: the description stays out of
|
||||
// it and is available through `{content}`.
|
||||
assert!(!fetched.caption.contains("一行说明"), "{}", fetched.caption);
|
||||
assert_eq!(
|
||||
fetched.render_fields().unwrap().3,
|
||||
"一行说明\n二行 链接 & 结尾"
|
||||
);
|
||||
assert!(
|
||||
fetched
|
||||
.caption_with("{title}: {content}")
|
||||
.ends_with("一行说明\n二行 链接 & 结尾")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn flatten_html_handles_common_markup() {
|
||||
assert_eq!(flatten_html(""), "");
|
||||
assert_eq!(flatten_html("plain"), "plain");
|
||||
assert_eq!(flatten_html("a<br />b<br/>c<br>d"), "a\nb\nc\nd");
|
||||
// A paragraph break is a blank line, exactly like `<br /><br />` —
|
||||
// writing it as one newline would flatten the author's paragraphs.
|
||||
assert_eq!(flatten_html("<p>one</p><p>two</p>"), "one\n\ntwo");
|
||||
assert_eq!(flatten_html("a & b <c>"), "a & b <c>");
|
||||
// Nothing to strip: angle brackets that are not a tag survive.
|
||||
assert_eq!(flatten_html("2 < 3"), "2 < 3");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pattern_matches_all_forms() {
|
||||
let cases = [
|
||||
@@ -232,6 +410,43 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_retryable_classifies_transient_and_permanent() {
|
||||
// Transient: network errors, explicit transient, pixiv 429/5xx.
|
||||
assert!(is_retryable(&FetchError::Transient("429".into())));
|
||||
assert!(is_retryable(&FetchError::Pixiv(PixivError::Status(429))));
|
||||
assert!(is_retryable(&FetchError::Pixiv(PixivError::Status(500))));
|
||||
assert!(is_retryable(&FetchError::Pixiv(PixivError::Status(503))));
|
||||
// Permanent: pixiv 4xx (bad/expired token, forbidden, not found),
|
||||
// api/auth errors, unparseable bodies, not-found/blocked/sensitive.
|
||||
assert!(!is_retryable(&FetchError::Pixiv(PixivError::Status(400))));
|
||||
assert!(!is_retryable(&FetchError::Pixiv(PixivError::Status(401))));
|
||||
assert!(!is_retryable(&FetchError::Pixiv(PixivError::Status(403))));
|
||||
assert!(!is_retryable(&FetchError::Pixiv(PixivError::Status(404))));
|
||||
assert!(!is_retryable(&FetchError::Pixiv(PixivError::Api(
|
||||
"invalid_grant".into()
|
||||
))));
|
||||
assert!(!is_retryable(&FetchError::Pixiv(PixivError::NoAuth)));
|
||||
let json_err = serde_json::from_str::<serde_json::Value>("x").unwrap_err();
|
||||
assert!(!is_retryable(&FetchError::Pixiv(PixivError::Json(
|
||||
json_err
|
||||
))));
|
||||
assert!(!is_retryable(&FetchError::NotFound));
|
||||
assert!(!is_retryable(&FetchError::Blocked));
|
||||
assert!(!is_retryable(&FetchError::Sensitive));
|
||||
assert!(!is_retryable(&FetchError::TooLarge));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn media_headers_adds_referer_only_for_pximg() {
|
||||
assert_eq!(
|
||||
media_headers("https://i.pximg.net/img-original/img/1.png"),
|
||||
Some(vec![("Referer", "https://www.pixiv.net/".to_string())])
|
||||
);
|
||||
assert_eq!(media_headers("https://www.pixiv.net/artworks/1"), None);
|
||||
assert_eq!(media_headers("https://x.com/u/status/1"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ugoira_yields_empty_media() {
|
||||
let v = illust_json(
|
||||
|
||||
@@ -3,4 +3,7 @@ mod interface;
|
||||
mod model;
|
||||
|
||||
pub use api::{PixivAPI, PixivError, disable, fetch, validate};
|
||||
pub use interface::{Illustration, PATTERN, enabled, fetch_from_url};
|
||||
pub use interface::{
|
||||
Illustration, PATTERN, PixivSite, cache_key, enabled, fetch_from_url, is_retryable,
|
||||
media_headers,
|
||||
};
|
||||
|
||||
@@ -6,6 +6,10 @@ use serde::Deserialize;
|
||||
pub struct IllustrationModel {
|
||||
pub id: u64,
|
||||
pub title: String,
|
||||
/// The artwork's description as the app API returns it — HTML in most
|
||||
/// works (`<br />`, `<a href>`, sometimes `<p>`), empty for many.
|
||||
#[serde(default)]
|
||||
pub caption: String,
|
||||
pub r#type: TypeModel,
|
||||
pub image_urls: ImageUrlsModel,
|
||||
pub user: UserInfoModel,
|
||||
|
||||
@@ -126,9 +126,16 @@ pub async fn fetch(id: &str) -> Result<Tweet, FetchError> {
|
||||
.header("referer", "https://x.com/")
|
||||
.send()
|
||||
.await?;
|
||||
if !response.status().is_success() {
|
||||
log::warn!("twitter auth fetch {id}: HTTP {}", response.status());
|
||||
return Err(FetchError::NotFound);
|
||||
// 404/410 = gone (permanent); 429/5xx = transient and retried by fetch.
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
log::warn!("twitter auth fetch {id}: HTTP {status}");
|
||||
return match status.as_u16() {
|
||||
404 | 410 => Err(FetchError::NotFound),
|
||||
_ => Err(FetchError::Transient(format!(
|
||||
"twitter auth status {status}"
|
||||
))),
|
||||
};
|
||||
}
|
||||
let text = response.text().await?;
|
||||
let json: Value = serde_json::from_str(&text)?;
|
||||
@@ -320,7 +327,8 @@ mod tests {
|
||||
"https://x.com/nsfw_author/status/2083868672721039569"
|
||||
);
|
||||
// The appended media short link (no URL-entity mapping) is stripped.
|
||||
assert_eq!(fetched.title, "nsfw content");
|
||||
assert_eq!(fetched.title, "");
|
||||
assert_eq!(fetched.content, "nsfw content");
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -1,10 +1,31 @@
|
||||
use super::model;
|
||||
use crate::media::Media;
|
||||
use crate::site::{FetchError, Fetched};
|
||||
use html_escape::encode_text;
|
||||
use crate::site::{FetchError, Fetched, Site, SiteFuture};
|
||||
use html_escape::{decode_html_entities, encode_double_quoted_attribute, encode_text};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
/// Registry entry for the twitter adapter (see [`crate::site::Site`]).
|
||||
pub struct TwitterSite;
|
||||
|
||||
impl Site for TwitterSite {
|
||||
fn id(&self) -> &'static str {
|
||||
"twitter"
|
||||
}
|
||||
|
||||
fn pattern(&self) -> &'static Regex {
|
||||
&PATTERN
|
||||
}
|
||||
|
||||
fn cache_key(&self, url: &str) -> Option<String> {
|
||||
cache_key(url)
|
||||
}
|
||||
|
||||
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched> {
|
||||
Box::pin(async move { fetch_from_url(url).await })
|
||||
}
|
||||
}
|
||||
|
||||
pub static PATTERN: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"^(?:https?://)?(?:www\.|mobile\.)?(?:x|twitter|fixvx|vxtwitter|fixupx|fxtwitter)\.com/[^.]+/status/(\d+)").unwrap()
|
||||
});
|
||||
@@ -29,13 +50,19 @@ pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> {
|
||||
if super::auth::enabled() {
|
||||
match super::auth::fetch(id).await {
|
||||
Ok(tweet) => Ok(tweet.into()),
|
||||
// The tweet is genuinely gone (deleted / suspended /
|
||||
// tombstoned): report it instead of degrading to an
|
||||
// empty result ("No media found"). Only unexpected
|
||||
// fallback failures (network, parse) keep the NSFW
|
||||
// placeholder.
|
||||
Err(FetchError::NotFound) => Err(FetchError::NotFound),
|
||||
Err(e) => {
|
||||
log::warn!("twitter auth fallback failed for {id}: {e}");
|
||||
Ok(empty_fetched(url))
|
||||
}
|
||||
}
|
||||
} else {
|
||||
log::info!("tweet {id} is sensitive; set TWITTER_AUTH_TOKEN to fetch NSFW media");
|
||||
log::debug!("tweet {id} is sensitive; set TWITTER_AUTH_TOKEN to fetch NSFW media");
|
||||
Ok(empty_fetched(url))
|
||||
}
|
||||
}
|
||||
@@ -43,22 +70,47 @@ pub async fn fetch_from_url(url: &str) -> Result<Fetched, FetchError> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Cache key for a twitter URL: `"twitter:<id>"`. The prefix is the site id
|
||||
/// used for caption-format lookup and link-cache keys.
|
||||
pub fn cache_key(url: &str) -> Option<String> {
|
||||
PATTERN
|
||||
.captures(url)
|
||||
.map(|caps| format!("twitter:{}", &caps[1]))
|
||||
}
|
||||
|
||||
/// Twitter's fetch-retry policy: transient classes only. Not-found, blocked,
|
||||
/// sensitive (NSFW withholding) and parse failures are permanent — retrying
|
||||
/// them only wastes attempts against the syndication endpoint.
|
||||
pub fn is_retryable(err: &FetchError) -> bool {
|
||||
matches!(err, FetchError::Http(_) | FetchError::Transient(_))
|
||||
}
|
||||
|
||||
/// twimg URLs need no extra headers (no hotlink protection).
|
||||
pub fn media_headers(_url: &str) -> Option<Vec<(&'static str, String)>> {
|
||||
None
|
||||
}
|
||||
|
||||
/// A Fetched with no media for withheld tweets: the bot replies
|
||||
/// "No media found" and moves on instead of erroring.
|
||||
fn empty_fetched(url: &str) -> Fetched {
|
||||
Fetched {
|
||||
source_url: url.to_string(),
|
||||
caption: url.to_string(),
|
||||
// The raw user-supplied URL goes into an HTML caption; escape it so
|
||||
// crafted links cannot break the parse (Telegram 400).
|
||||
caption: encode_text(url).into_owned(),
|
||||
title: String::new(),
|
||||
content: String::new(),
|
||||
media: vec![],
|
||||
sensitive: true,
|
||||
site_id: "twitter",
|
||||
render_data: None,
|
||||
_keep_alive: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Fetches a tweet from the syndication endpoint. Deleted/blocked tweets
|
||||
/// surface as `FetchError::NotFound`.
|
||||
/// surface as `FetchError::NotFound`; withheld content (empty tombstone,
|
||||
/// age-restricted) as `FetchError::Sensitive`.
|
||||
pub async fn fetch(id: &str) -> Result<Tweet, FetchError> {
|
||||
let id_num = id.parse::<u64>().map_err(|_| FetchError::NotFound)?;
|
||||
let response = crate::site::CLIENT
|
||||
@@ -68,27 +120,55 @@ pub async fn fetch(id: &str) -> Result<Tweet, FetchError> {
|
||||
))
|
||||
.send()
|
||||
.await?;
|
||||
if !response.status().is_success() {
|
||||
return Err(FetchError::NotFound);
|
||||
// 404/410 = gone (permanent); 429/5xx = transient and retried by fetch.
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
return match status.as_u16() {
|
||||
404 | 410 => Err(FetchError::NotFound),
|
||||
_ => Err(FetchError::Transient(format!("twitter status {status}"))),
|
||||
};
|
||||
}
|
||||
let text = response.text().await?;
|
||||
// Deleted tweets answer with {"errors": [...]} instead of a tweet.
|
||||
if serde_json::from_str::<serde_json::Value>(&text)
|
||||
.map(|v| v.get("errors").is_some())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
// Classify before parsing the tweet (see [`parse_syndication_body`]).
|
||||
parse_syndication_body(&text)?;
|
||||
Tweet::from_syndication_json(&text).map_err(FetchError::Json)
|
||||
}
|
||||
|
||||
/// Parses and classifies a syndication response body. `Ok` means the body is
|
||||
/// a real tweet payload; `Err` carries the permanent error class:
|
||||
/// - `NotFound`: an `errors` array (deleted/blocked) or a `TweetTombstone`
|
||||
/// **with a reason** — "This Post was deleted by the Post author." /
|
||||
/// "This Post is from a suspended account." (the tweet is gone).
|
||||
/// - `Sensitive`: content withheld **without a deletion reason** — the empty
|
||||
/// `{}` shape or an *empty* `TweetTombstone` (`{"__typename":
|
||||
/// "TweetTombstone","tombstone":{}}`). Live tweets in restricted contexts
|
||||
/// surface this way; treating them as deleted is a regression (a normal
|
||||
/// tweet must not report "deleted"). Age-restricted tombstones route here
|
||||
/// too so the logged-in GraphQL fallback can fetch the real tweet.
|
||||
/// - `Json`: an unparseable body.
|
||||
fn parse_syndication_body(text: &str) -> Result<serde_json::Value, FetchError> {
|
||||
let body: serde_json::Value = serde_json::from_str(text)?;
|
||||
if body.get("errors").is_some() {
|
||||
return Err(FetchError::NotFound);
|
||||
}
|
||||
// NSFW / age-restricted tweets exist but are served as an empty `{}` —
|
||||
// they surface as FetchError::Sensitive so the caller can retry as a
|
||||
// logged-in user.
|
||||
if serde_json::from_str::<serde_json::Value>(&text)
|
||||
.map(|v| v.get("id_str").is_none())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
if let Some(tombstone) = body.get("tombstone") {
|
||||
// Only a tombstone with an explicit reason means the tweet is gone;
|
||||
// a missing reason (empty `tombstone: {}`) or an age-restricted
|
||||
// reason means the tweet exists but is withheld.
|
||||
let reason = tombstone
|
||||
.get("text")
|
||||
.and_then(|t| t.get("text"))
|
||||
.and_then(|t| t.as_str())
|
||||
.unwrap_or("");
|
||||
if reason.is_empty() || reason.to_ascii_lowercase().contains("age-restricted") {
|
||||
return Err(FetchError::Sensitive);
|
||||
}
|
||||
return Err(FetchError::NotFound);
|
||||
}
|
||||
if body.get("id_str").is_none() {
|
||||
return Err(FetchError::Sensitive);
|
||||
}
|
||||
Tweet::from_syndication_json(&text).map_err(FetchError::Json)
|
||||
Ok(body)
|
||||
}
|
||||
|
||||
/// The syndication token: JS `((id / 1e15) * PI).toString(36)` (the
|
||||
@@ -146,8 +226,8 @@ impl Tweet {
|
||||
pub fn caption(&self) -> String {
|
||||
format!(
|
||||
"{url}\n<a href=\"{author_url}\">{author}</a>: {text}",
|
||||
url = self.url(),
|
||||
author_url = self.author_url(),
|
||||
url = encode_double_quoted_attribute(&self.url()),
|
||||
author_url = encode_double_quoted_attribute(&self.author_url()),
|
||||
author = encode_text(&self.author),
|
||||
text = encode_text(&self.text),
|
||||
)
|
||||
@@ -160,9 +240,17 @@ impl Tweet {
|
||||
// strip the appended media short link, mirroring FxEmbed's linkFixer
|
||||
// (no display_text_range arithmetic — see expand_links).
|
||||
let text = expand_links(&json.text, &json.entities.urls);
|
||||
// Twitter APIs (syndication AND GraphQL full_text) return the text
|
||||
// pre-escaped for HTML (`>` `<` `&` `'` …): decode it so
|
||||
// the stored text is raw. The caption's own escaping then produces
|
||||
// the rendered form exactly once — without this, `>^ω^<` would
|
||||
// be double-escaped to `&gt;^ω^&lt;` and the sent message
|
||||
// would show literal `>^ω^<`.
|
||||
let text = decode_html_entities(&text).into_owned();
|
||||
// `name` is the display name, `screen_name` the handle (Python's
|
||||
// vxtwitter mapping: author = display name, author_id = handle).
|
||||
let author = json.user.name;
|
||||
// Display names can carry the same pre-escaped entities.
|
||||
let author = decode_html_entities(&json.user.name).into_owned();
|
||||
let author_id = json.user.screen_name;
|
||||
let mut media = vec![];
|
||||
for item in json.media_details {
|
||||
@@ -265,19 +353,23 @@ impl From<Tweet> for Fetched {
|
||||
fn from(tweet: Tweet) -> Self {
|
||||
let url = tweet.url();
|
||||
let author_url = tweet.author_url();
|
||||
// A tweet has no title: its text is all content.
|
||||
let render_data = Some(crate::site::RenderData {
|
||||
url: url.clone(),
|
||||
author: encode_text(&tweet.author).into_owned(),
|
||||
author_url: author_url.clone(),
|
||||
title: encode_text(&tweet.text).into_owned(),
|
||||
title: String::new(),
|
||||
content: encode_text(&tweet.text).into_owned(),
|
||||
tags: String::new(),
|
||||
});
|
||||
Fetched {
|
||||
source_url: url,
|
||||
caption: tweet.caption(),
|
||||
title: tweet.text.clone(),
|
||||
title: String::new(),
|
||||
content: tweet.text.clone(),
|
||||
media: tweet.media,
|
||||
sensitive: tweet.sensitive,
|
||||
site_id: "twitter",
|
||||
render_data,
|
||||
_keep_alive: None,
|
||||
}
|
||||
@@ -329,6 +421,65 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn syndication_text_is_unescaped_before_storing() {
|
||||
// Real API shape: the text arrives pre-escaped for HTML — e.g. the
|
||||
// tweet `>^ω^<` comes back as `>^ω^<` (fxtwitter's raw_text for
|
||||
// 2060196388252827954) and apostrophes as `'`. Storing it raw and
|
||||
// escaping once at caption build avoids the double-escape that would
|
||||
// show literal `>`/`<`/`&` in the sent message.
|
||||
let raw = serde_json::json!({
|
||||
"__typename": "Tweet",
|
||||
"id_str": "1",
|
||||
"text": ">^ω^< & more 'quoted' https://t.co/abc123",
|
||||
"user": { "name": "O'Brien", "screen_name": "h" },
|
||||
"entities": { "urls": [] },
|
||||
"mediaDetails": []
|
||||
});
|
||||
let tweet = Tweet::from_syndication_json(&raw.to_string()).unwrap();
|
||||
// The appended media short link is stripped, then entities decoded.
|
||||
assert_eq!(tweet.text, ">^ω^< & more 'quoted'");
|
||||
assert_eq!(tweet.author, "O'Brien");
|
||||
let fetched: Fetched = tweet.into();
|
||||
assert_eq!(fetched.title, "");
|
||||
assert_eq!(fetched.content, ">^ω^< & more 'quoted'");
|
||||
// The caption escapes the raw text exactly once (encode_text covers
|
||||
// & < >; apostrophes stay literal — they are harmless in text).
|
||||
assert!(
|
||||
fetched.caption.contains(">^ω^< & more 'quoted'"),
|
||||
"caption: {}",
|
||||
fetched.caption
|
||||
);
|
||||
assert!(
|
||||
!fetched.caption.contains("&gt;"),
|
||||
"double-escaped text: {}",
|
||||
fetched.caption
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cache_key_prefixes_tweet_id() {
|
||||
assert_eq!(
|
||||
cache_key("https://x.com/user/status/1234567890"),
|
||||
Some("twitter:1234567890".into())
|
||||
);
|
||||
assert_eq!(cache_key("https://example.com/1"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_retryable_classifies_transient_and_permanent() {
|
||||
// Transient: network errors and explicit transient statuses (the
|
||||
// `Http` arm shares this match arm with `Transient`).
|
||||
assert!(is_retryable(&FetchError::Transient("429".into())));
|
||||
// Permanent: gone, blocked, withheld, oversized, unparseable.
|
||||
assert!(!is_retryable(&FetchError::NotFound));
|
||||
assert!(!is_retryable(&FetchError::Blocked));
|
||||
assert!(!is_retryable(&FetchError::Sensitive));
|
||||
assert!(!is_retryable(&FetchError::TooLarge));
|
||||
let json_err = serde_json::from_str::<serde_json::Value>("x").unwrap_err();
|
||||
assert!(!is_retryable(&FetchError::Json(json_err)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn syndication_json_converts_to_fetched() {
|
||||
let raw = fixture(serde_json::json!([
|
||||
@@ -350,7 +501,8 @@ mod tests {
|
||||
fetched.source_url,
|
||||
"https://x.com/author_handle/status/861627479294746624"
|
||||
);
|
||||
assert_eq!(fetched.title, "a & b <c>");
|
||||
assert_eq!(fetched.title, "");
|
||||
assert_eq!(fetched.content, "a & b <c>");
|
||||
assert!(fetched.sensitive);
|
||||
assert_eq!(fetched.media.len(), 2);
|
||||
match &fetched.media[0] {
|
||||
@@ -565,19 +717,93 @@ mod tests {
|
||||
assert!(token.starts_with("236.v"), "got {token}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn syndication_tombstone_maps_to_not_found() {
|
||||
// Deleted tweets answer HTTP 200 with a TweetTombstone carrying a
|
||||
// reason (no `errors`, no `id_str`); they must not fall through to
|
||||
// Sensitive, which would make the bot reply "No media found" for a
|
||||
// deleted tweet.
|
||||
let raw = serde_json::json!({
|
||||
"__typename": "TweetTombstone",
|
||||
"tombstone": {
|
||||
"text": { "rtl": false, "text": "This Post was deleted by the Post author. Learn more" }
|
||||
}
|
||||
});
|
||||
assert!(matches!(
|
||||
parse_syndication_body(&raw.to_string()),
|
||||
Err(FetchError::NotFound)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn syndication_empty_tombstone_maps_to_sensitive() {
|
||||
// Regression: live tweets in restricted contexts answer with an
|
||||
// EMPTY tombstone (`{"__typename":"TweetTombstone","tombstone":{}}`)
|
||||
// — no deletion reason. They must not be reported as deleted.
|
||||
let raw = serde_json::json!({ "__typename": "TweetTombstone", "tombstone": {} });
|
||||
assert!(matches!(
|
||||
parse_syndication_body(&raw.to_string()),
|
||||
Err(FetchError::Sensitive)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn syndication_age_restricted_tombstone_maps_to_sensitive() {
|
||||
// An age-restricted tombstone withholds a live tweet; route it to
|
||||
// the logged-in fallback instead of reporting it as gone.
|
||||
let raw = serde_json::json!({
|
||||
"__typename": "TweetTombstone",
|
||||
"tombstone": {
|
||||
"text": { "rtl": false, "text": "Age-restricted adult content" }
|
||||
}
|
||||
});
|
||||
assert!(matches!(
|
||||
parse_syndication_body(&raw.to_string()),
|
||||
Err(FetchError::Sensitive)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn syndication_errors_maps_to_not_found() {
|
||||
// The classic gone shape: {"errors": [...]}.
|
||||
let raw = serde_json::json!({ "errors": [{ "message": "Couldn't find Tweet" }] });
|
||||
assert!(matches!(
|
||||
parse_syndication_body(&raw.to_string()),
|
||||
Err(FetchError::NotFound)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn syndication_empty_object_maps_to_sensitive() {
|
||||
// NSFW / age-restricted withholding: an empty `{}`.
|
||||
assert!(matches!(
|
||||
parse_syndication_body("{}"),
|
||||
Err(FetchError::Sensitive)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn syndication_tweet_body_passes() {
|
||||
let raw = fixture(serde_json::json!([]));
|
||||
assert!(parse_syndication_body(&raw.to_string()).is_ok());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "live network: requires outbound HTTPS to cdn.syndication.twimg.com"]
|
||||
async fn live_fetch_with_photos() {
|
||||
let fetched = fetch("861627479294746624").await.unwrap();
|
||||
assert_eq!(fetched.media.len(), 4);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "live network: requires outbound HTTPS to cdn.syndication.twimg.com"]
|
||||
async fn live_fetch_text_only() {
|
||||
let fetched = fetch("1992471125734142256").await.unwrap();
|
||||
assert!(fetched.media.is_empty());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "live network: requires outbound HTTPS to cdn.syndication.twimg.com"]
|
||||
async fn live_fetch_deleted_tweet_is_not_found() {
|
||||
// Deleted tweet: the syndication endpoint answers with errors.
|
||||
let result = fetch("0").await;
|
||||
@@ -586,4 +812,30 @@ mod tests {
|
||||
"got {result:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "live network: requires outbound HTTPS to cdn.syndication.twimg.com"]
|
||||
async fn live_fetch_tombstone_deleted_tweet_is_not_found() {
|
||||
// Regression: a real deleted tweet answering with a TweetTombstone
|
||||
// (HTTP 200, no errors/id_str) used to surface as Sensitive and
|
||||
// degrade to an empty result ("No media found").
|
||||
let result = fetch("2085948045967986859").await;
|
||||
assert!(
|
||||
matches!(result, Err(FetchError::NotFound)),
|
||||
"got {result:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "live network: requires outbound HTTPS to cdn.syndication.twimg.com"]
|
||||
async fn live_fetch_empty_tombstone_is_sensitive() {
|
||||
// Regression: a LIVE tweet (verified via a third-party API) answers
|
||||
// syndication with an empty TweetTombstone; it must surface as
|
||||
// Sensitive (withheld), never as NotFound (deleted).
|
||||
let result = fetch("2087851366253555752").await;
|
||||
assert!(
|
||||
matches!(result, Err(FetchError::Sensitive)),
|
||||
"got {result:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,4 +2,6 @@ mod auth;
|
||||
mod interface;
|
||||
mod model;
|
||||
|
||||
pub use interface::{PATTERN, Tweet, enabled, fetch_from_url};
|
||||
pub use interface::{
|
||||
PATTERN, Tweet, TwitterSite, cache_key, enabled, fetch_from_url, is_retryable, media_headers,
|
||||
};
|
||||
|
||||
@@ -1,25 +1,28 @@
|
||||
[package]
|
||||
name = "xmedia-bot"
|
||||
version = "1.0.8"
|
||||
version = "1.7.0"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
teloxide = { version = "0.17", features = ["webhooks-axum", "macros"] }
|
||||
tokio = { version = "1.40", features = ["rt-multi-thread", "macros"] }
|
||||
teloxide = { version = "0.17", default-features = false, features = ["webhooks-axum", "macros", "rustls", "ctrlc_handler"] }
|
||||
tokio = { version = "1.40", features = ["rt-multi-thread", "macros", "time", "sync"] }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
log = "0.4"
|
||||
pretty_env_logger = "0.5"
|
||||
dotenv = "0.15"
|
||||
url = "2.5.2"
|
||||
regex = "1.12"
|
||||
html-escape = "0.2"
|
||||
rusqlite = { version = "0.32", features = ["bundled"] }
|
||||
rand = "0.8"
|
||||
rusqlite = { version = "0.40", features = ["bundled"] }
|
||||
rand = "0.10"
|
||||
tempfile = "3"
|
||||
parking_lot = "0.12"
|
||||
bytes = "1"
|
||||
png = "0.18"
|
||||
zune-jpeg = "0.5"
|
||||
fast_image_resize = "6"
|
||||
jpeg-encoder = "0.7"
|
||||
x-media = { path = "../x-media" }
|
||||
|
||||
[dev-dependencies]
|
||||
tokio = { version = "1.40", features = ["test-util"] }
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
//! Central env handling. The only other places that read env are
|
||||
//! `Bot::from_env` (TELOXIDE_TOKEN) and x-media (PIXIV_REFRESH_TOKEN).
|
||||
//! `Bot::from_env` (TELOXIDE_TOKEN) and x-media (PIXIV_REFRESH_TOKEN,
|
||||
//! TWITTER_AUTH_TOKEN, BILIBILI_COOKIE).
|
||||
|
||||
use std::env;
|
||||
use std::net::IpAddr;
|
||||
@@ -12,6 +13,10 @@ pub struct Config {
|
||||
pub edit_message_ttl: Duration,
|
||||
/// LINK_CACHE_TTL_SECONDS, default 604800 (7 days).
|
||||
pub link_cache_ttl: Duration,
|
||||
/// CAPTION_QUOTE_TEXT_CHARS, default 200: a post whose text (title plus
|
||||
/// content) is at least this many characters gets that text wrapped in an
|
||||
/// expandable blockquote inside its caption. `0` disables the wrap.
|
||||
pub caption_quote_text_chars: usize,
|
||||
// Webhook settings (moved out of main; names/defaults unchanged).
|
||||
pub webhook_enabled: bool,
|
||||
pub webhook_url: Option<url::Url>,
|
||||
@@ -23,32 +28,65 @@ pub struct Config {
|
||||
|
||||
impl Config {
|
||||
pub fn load() -> Config {
|
||||
let admin_ids = env::var("BOT_ADMIN")
|
||||
.ok()
|
||||
.map(|s| {
|
||||
s.split(',')
|
||||
.filter_map(|part| part.trim().parse::<i64>().ok())
|
||||
// Fail-fast helpers: a misspelled value must not silently fall back
|
||||
// to a default and run with different behavior than the operator
|
||||
// intended — log a loud warning naming the variable instead.
|
||||
fn parse_u64(name: &str, default: u64) -> u64 {
|
||||
match env::var(name) {
|
||||
Ok(v) => v.parse::<u64>().unwrap_or_else(|_| {
|
||||
log::warn!("invalid {name}={v:?}; using default {default}");
|
||||
default
|
||||
}),
|
||||
Err(_) => default,
|
||||
}
|
||||
}
|
||||
|
||||
let admin_ids = match env::var("BOT_ADMIN") {
|
||||
Ok(s) => {
|
||||
let (ids, bad): (Vec<_>, Vec<_>) = s
|
||||
.split(',')
|
||||
.map(str::trim)
|
||||
.filter(|part| !part.is_empty())
|
||||
.partition(|part| part.parse::<i64>().is_ok());
|
||||
if !bad.is_empty() {
|
||||
log::warn!("BOT_ADMIN: ignoring non-numeric ids: {bad:?}");
|
||||
}
|
||||
ids.into_iter()
|
||||
.filter_map(|p| p.parse::<i64>().ok())
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default();
|
||||
}
|
||||
Err(_) => Vec::new(),
|
||||
};
|
||||
|
||||
let edit_message_ttl = env::var("EDIT_MESSAGE_TTL_SECONDS")
|
||||
.ok()
|
||||
.and_then(|s| s.parse::<u64>().ok())
|
||||
.map(Duration::from_secs)
|
||||
.unwrap_or(Duration::from_secs(86400));
|
||||
|
||||
let link_cache_ttl = env::var("LINK_CACHE_TTL_SECONDS")
|
||||
.ok()
|
||||
.and_then(|s| s.parse::<u64>().ok())
|
||||
.map(Duration::from_secs)
|
||||
.unwrap_or(Duration::from_secs(7 * 24 * 3600));
|
||||
let edit_message_ttl =
|
||||
Duration::from_secs(parse_u64("EDIT_MESSAGE_TTL_SECONDS", 24 * 3600));
|
||||
let link_cache_ttl =
|
||||
Duration::from_secs(parse_u64("LINK_CACHE_TTL_SECONDS", 7 * 24 * 3600));
|
||||
let caption_quote_text_chars = parse_u64("CAPTION_QUOTE_TEXT_CHARS", 200) as usize;
|
||||
|
||||
let webhook_enabled = env::var("WEBHOOK")
|
||||
.is_ok_and(|v| matches!(v.to_lowercase().as_str(), "true" | "yes" | "1"));
|
||||
let webhook_url = env::var("WEBHOOK_URL").ok().and_then(|s| s.parse().ok());
|
||||
let webhook_listen = env::var("WEBHOOK_LISTEN").ok().and_then(|s| s.parse().ok());
|
||||
let webhook_port = env::var("WEBHOOK_PORT").ok().and_then(|s| s.parse().ok());
|
||||
// The webhook settings are consumed by `.expect()` in main when
|
||||
// WEBHOOK=true, so an unparseable value fails fast at startup with a
|
||||
// clear message; still log here for the WEBHOOK=false case.
|
||||
let webhook_url = env::var("WEBHOOK_URL").ok().and_then(|s| {
|
||||
s.parse::<url::Url>().ok().or_else(|| {
|
||||
log::warn!("invalid WEBHOOK_URL={s:?}");
|
||||
None
|
||||
})
|
||||
});
|
||||
let webhook_listen = env::var("WEBHOOK_LISTEN").ok().and_then(|s| {
|
||||
s.parse::<IpAddr>().ok().or_else(|| {
|
||||
log::warn!("invalid WEBHOOK_LISTEN={s:?}");
|
||||
None
|
||||
})
|
||||
});
|
||||
let webhook_port = env::var("WEBHOOK_PORT").ok().and_then(|s| {
|
||||
s.parse::<u16>().ok().or_else(|| {
|
||||
log::warn!("invalid WEBHOOK_PORT={s:?}");
|
||||
None
|
||||
})
|
||||
});
|
||||
// Empty strings count as unset (e.g. `-e WEBHOOK_CERT=` to disable a
|
||||
// value that would otherwise come from `.env`).
|
||||
let webhook_cert = env::var("WEBHOOK_CERT").ok().filter(|s| !s.is_empty());
|
||||
@@ -60,6 +98,7 @@ impl Config {
|
||||
admin_ids,
|
||||
edit_message_ttl,
|
||||
link_cache_ttl,
|
||||
caption_quote_text_chars,
|
||||
webhook_enabled,
|
||||
webhook_url,
|
||||
webhook_listen,
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
//! Runtime context: the collaborators a handler needs, injected as one struct
|
||||
//! so tests can substitute a scripted sender and tempdir-backed stores.
|
||||
//!
|
||||
//! The production context is assembled from the process-wide statics
|
||||
//! ([`AppContext::from_statics`]); the spawned worker closures hold
|
||||
//! [`CONTEXT`], which is `'static` for that reason.
|
||||
|
||||
use crate::config::Config;
|
||||
use crate::handlers::{CHAT_STORE, CONFIG, LINK_CACHE, TASK_QUEUE};
|
||||
use crate::link_cache::LinkCache;
|
||||
use crate::media_sender::MediaSender;
|
||||
use crate::queue::PersistentTaskQueue;
|
||||
use crate::send::BOT;
|
||||
use crate::state::ChatStore;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub struct AppContext<'a> {
|
||||
pub sender: &'a dyn MediaSender,
|
||||
pub chat_store: &'a ChatStore,
|
||||
pub task_queue: &'a PersistentTaskQueue,
|
||||
pub link_cache: &'a LinkCache,
|
||||
pub config: &'a Config,
|
||||
}
|
||||
|
||||
impl<'a> AppContext<'a> {
|
||||
/// The stores are the process-wide statics; `sender` is whatever the caller
|
||||
/// was handed (the dispatcher's `Bot` clone for update handlers, the shared
|
||||
/// queue `Bot` for the worker loops). Update handlers build their own
|
||||
/// context from the `Bot` they received so the same code path works with an
|
||||
/// injected mock in tests.
|
||||
pub fn from_statics(sender: &'a dyn MediaSender) -> AppContext<'a> {
|
||||
AppContext {
|
||||
sender,
|
||||
chat_store: &CHAT_STORE,
|
||||
task_queue: &TASK_QUEUE,
|
||||
link_cache: &LINK_CACHE,
|
||||
config: &CONFIG,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The URL/queue workers' context: `'static` because `tokio::spawn`ed closures
|
||||
/// and the queue's handler type require it.
|
||||
pub static CONTEXT: LazyLock<AppContext<'static>> =
|
||||
LazyLock::new(|| AppContext::from_statics(&*BOT));
|
||||
|
||||
/// Test support: a tempdir-backed set of stores plus the context borrowing
|
||||
/// them, so a handler test needs one line of setup.
|
||||
#[cfg(test)]
|
||||
pub(crate) mod test_support {
|
||||
use super::*;
|
||||
use std::sync::Arc;
|
||||
|
||||
pub(crate) struct TestStores {
|
||||
_dir: tempfile::TempDir,
|
||||
pool: Arc<crate::db::DbPool>,
|
||||
chat_store: ChatStore,
|
||||
task_queue: PersistentTaskQueue,
|
||||
link_cache: LinkCache,
|
||||
config: Config,
|
||||
}
|
||||
|
||||
impl TestStores {
|
||||
pub(crate) fn new() -> Self {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pool = crate::db::open_store(dir.path().join("ctx.db").to_str().unwrap()).unwrap();
|
||||
TestStores {
|
||||
_dir: dir,
|
||||
chat_store: ChatStore::new(Arc::clone(&pool)),
|
||||
task_queue: PersistentTaskQueue::new(Arc::clone(&pool)),
|
||||
link_cache: LinkCache::new(Arc::clone(&pool)),
|
||||
config: Config::load(),
|
||||
pool,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn ctx<'a>(&'a self, sender: &'a dyn MediaSender) -> AppContext<'a> {
|
||||
AppContext {
|
||||
sender,
|
||||
chat_store: &self.chat_store,
|
||||
task_queue: &self.task_queue,
|
||||
link_cache: &self.link_cache,
|
||||
config: &self.config,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn chat_store(&self) -> &ChatStore {
|
||||
&self.chat_store
|
||||
}
|
||||
|
||||
/// The parsed config, mutable so a test can pin a knob (e.g. the
|
||||
/// caption-quote threshold) instead of depending on the environment.
|
||||
pub(crate) fn config_mut(&mut self) -> &mut Config {
|
||||
&mut self.config
|
||||
}
|
||||
|
||||
pub(crate) fn link_cache(&self) -> &LinkCache {
|
||||
&self.link_cache
|
||||
}
|
||||
|
||||
/// Rows persisted in the task queue: what "queued for retry" looks like
|
||||
/// from the outside.
|
||||
pub(crate) async fn queued_tasks(&self) -> i64 {
|
||||
let pool = Arc::clone(&self.pool);
|
||||
pool.with_conn(|conn| {
|
||||
conn.query_row("SELECT COUNT(*) FROM tasks", [], |row| row.get(0))
|
||||
})
|
||||
.await
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// The single queued task payload, for asserting what was rescheduled.
|
||||
pub(crate) async fn queued_payload(&self) -> serde_json::Value {
|
||||
let pool = Arc::clone(&self.pool);
|
||||
let payload: String = pool
|
||||
.with_conn(|conn| {
|
||||
conn.query_row("SELECT payload FROM tasks LIMIT 1", [], |row| row.get(0))
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
serde_json::from_str(&payload).unwrap()
|
||||
}
|
||||
}
|
||||
}
|
||||
+152
-21
@@ -2,36 +2,167 @@
|
||||
//! (`tasks` in queue.rs, `chat_state` in state.rs, `link_cache` in
|
||||
//! link_cache.rs).
|
||||
//!
|
||||
//! Every operation opens its own short-lived connection with a busy timeout:
|
||||
//! handler tasks enqueue while workers lease/update rows concurrently, and
|
||||
//! without the timeout a concurrent write fails immediately with SQLITE_BUSY
|
||||
//! and the operation is lost. All I/O runs inside `spawn_blocking` via
|
||||
//! [`with_conn`] — rusqlite connections are not Send-friendly to hold across
|
||||
//! an await point, and blocking the async executor stalls every handler.
|
||||
//! All I/O runs inside `spawn_blocking` via [`DbPool::with_conn`] — rusqlite
|
||||
//! connections are not Send-friendly to hold across an await point, and
|
||||
//! blocking the async executor stalls every handler. Connections are reused
|
||||
//! through a small per-store pool instead of opening a fresh connection per
|
||||
//! operation: WAL lets readers run alongside writer leases, and the pool's
|
||||
//! semaphore bounds how many DB operations run concurrently, giving natural
|
||||
//! backpressure on hot paths (every message / URL / callback touches
|
||||
//! chat_state or the link cache).
|
||||
|
||||
use parking_lot::Mutex;
|
||||
use rusqlite::Connection;
|
||||
use std::sync::Arc;
|
||||
use std::time::Duration;
|
||||
|
||||
/// Upper bound on pooled (reused) connections and on concurrent DB
|
||||
/// operations per store. Small on purpose: the queue's `BEGIN IMMEDIATE`
|
||||
/// leases serialize writes anyway, and WAL readers rarely need more.
|
||||
const POOL_SIZE: usize = 4;
|
||||
|
||||
/// A tiny connection pool for one SQLite file. Connections are checked out
|
||||
/// on a blocking thread and returned afterwards; `acquire` opens a new
|
||||
/// connection only when the idle list is empty, so the steady-state cost of
|
||||
/// an operation is a list pop instead of a fresh open (+ busy timeout + WAL
|
||||
/// pragma). The semaphore caps the number of concurrent operations, so a
|
||||
/// burst of handlers queues up instead of opening unbounded connections.
|
||||
pub struct DbPool {
|
||||
// Arc so [`DbPool::with_conn`] can hand an owned handle to
|
||||
// `spawn_blocking` without borrowing across the await point.
|
||||
inner: Arc<PoolInner>,
|
||||
}
|
||||
|
||||
struct PoolInner {
|
||||
path: String,
|
||||
permits: tokio::sync::Semaphore,
|
||||
idle: Mutex<Vec<Connection>>,
|
||||
}
|
||||
|
||||
impl DbPool {
|
||||
pub fn new(path: &str) -> Self {
|
||||
DbPool {
|
||||
inner: Arc::new(PoolInner {
|
||||
path: path.to_string(),
|
||||
permits: tokio::sync::Semaphore::new(POOL_SIZE),
|
||||
idle: Mutex::new(Vec::new()),
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
/// Runs `f` against a pooled connection on a blocking thread, returning
|
||||
/// the closure's result. Owns the semaphore + `spawn_blocking` +
|
||||
/// `expect` ceremony shared by every table access; the caller maps
|
||||
/// errors to its own log line.
|
||||
pub async fn with_conn<T, F>(&self, f: F) -> rusqlite::Result<T>
|
||||
where
|
||||
T: Send + 'static,
|
||||
F: FnOnce(&mut Connection) -> rusqlite::Result<T> + Send + 'static,
|
||||
{
|
||||
let _permit = self
|
||||
.inner
|
||||
.permits
|
||||
.acquire()
|
||||
.await
|
||||
.expect("db pool semaphore closed");
|
||||
let inner = Arc::clone(&self.inner);
|
||||
tokio::task::spawn_blocking(move || {
|
||||
let mut conn = inner.acquire()?;
|
||||
let result = f(&mut conn);
|
||||
inner.release(conn);
|
||||
result
|
||||
})
|
||||
.await
|
||||
.expect("db worker panicked")
|
||||
}
|
||||
|
||||
/// The database file this pool serves (used by tests that need a raw
|
||||
/// connection, e.g. to seed rows directly).
|
||||
#[cfg(test)]
|
||||
pub fn path(&self) -> &str {
|
||||
&self.inner.path
|
||||
}
|
||||
}
|
||||
|
||||
impl PoolInner {
|
||||
/// Reuses an idle connection or opens a fresh one.
|
||||
fn acquire(&self) -> rusqlite::Result<Connection> {
|
||||
if let Some(conn) = self.idle.lock().pop() {
|
||||
return Ok(conn);
|
||||
}
|
||||
open_db(&self.path)
|
||||
}
|
||||
|
||||
/// Returns a connection to the pool (dropped when the pool is full).
|
||||
fn release(&self, conn: Connection) {
|
||||
let mut idle = self.idle.lock();
|
||||
if idle.len() < POOL_SIZE {
|
||||
idle.push(conn);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Opens the shared DB with a busy timeout.
|
||||
pub fn open_db(path: &str) -> rusqlite::Result<Connection> {
|
||||
let conn = Connection::open(path)?;
|
||||
conn.busy_timeout(Duration::from_secs(5))?;
|
||||
// WAL lets readers run alongside writer leases instead of blocking on
|
||||
// the rollback journal; the mode persists in the DB header, so the
|
||||
// idempotent pragma here and in ensure_schema only needs to win once.
|
||||
conn.pragma_update(None, "journal_mode", "WAL")?;
|
||||
Ok(conn)
|
||||
}
|
||||
|
||||
/// Runs `f` against a fresh connection on a blocking thread, returning the
|
||||
/// closure's result. Owns the `spawn_blocking` + `expect` ceremony shared by
|
||||
/// every table access; the caller maps errors to its own log line.
|
||||
pub async fn with_conn<T, F>(path: &str, f: F) -> rusqlite::Result<T>
|
||||
where
|
||||
T: Send + 'static,
|
||||
F: FnOnce(&mut Connection) -> rusqlite::Result<T> + Send + 'static,
|
||||
{
|
||||
let path = path.to_string();
|
||||
tokio::task::spawn_blocking(move || {
|
||||
let mut conn = open_db(&path)?;
|
||||
f(&mut conn)
|
||||
})
|
||||
.await
|
||||
.expect("db worker panicked")
|
||||
/// Opens the shared DB file, runs the merged schema for all three tables and
|
||||
/// returns a pool for it. One call per process in production (the stores
|
||||
/// share the returned pool); tests call it per tempdir.
|
||||
pub fn open_store(path: &str) -> rusqlite::Result<Arc<DbPool>> {
|
||||
if let Some(parent) = std::path::Path::new(path).parent()
|
||||
&& !parent.as_os_str().is_empty()
|
||||
{
|
||||
std::fs::create_dir_all(parent).map_err(rusqlite_error)?;
|
||||
}
|
||||
let conn = open_db(path)?;
|
||||
schema_init(&conn)?;
|
||||
Ok(Arc::new(DbPool::new(path)))
|
||||
}
|
||||
|
||||
fn rusqlite_error(e: std::io::Error) -> rusqlite::Error {
|
||||
rusqlite::Error::ToSqlConversionFailure(Box::new(e))
|
||||
}
|
||||
|
||||
/// Creates the `tasks`, `chat_state` and `link_cache` tables (idempotent).
|
||||
/// The three stores used to own their own schema; keeping it in one place
|
||||
/// means one initialization for the whole database file.
|
||||
///
|
||||
/// ⚠️ Schema-change reminder (deferred, see `docs/architecture-refactor.md`
|
||||
/// §5): this is a plain `CREATE TABLE IF NOT EXISTS` with no versioning.
|
||||
/// Before any column/table change that must migrate existing databases, land
|
||||
/// the `PRAGMA user_version` migration chain first (`MIGRATIONS: &[&str]` +
|
||||
/// `migrate(conn)`), then restructure this function.
|
||||
pub fn schema_init(conn: &Connection) -> rusqlite::Result<()> {
|
||||
conn.execute_batch(
|
||||
"CREATE TABLE IF NOT EXISTS tasks (id TEXT PRIMARY KEY, payload TEXT NOT NULL, \
|
||||
run_after REAL NOT NULL, attempts INTEGER NOT NULL, status TEXT NOT NULL, \
|
||||
locked_until REAL NOT NULL, created_at REAL NOT NULL); \
|
||||
CREATE INDEX IF NOT EXISTS idx_tasks_pending ON tasks(status, run_after); \
|
||||
CREATE TABLE IF NOT EXISTS chat_state (chat_id TEXT PRIMARY KEY, payload TEXT NOT NULL); \
|
||||
CREATE TABLE IF NOT EXISTS link_cache (url TEXT PRIMARY KEY, payload TEXT NOT NULL, \
|
||||
created_at REAL NOT NULL);",
|
||||
)
|
||||
}
|
||||
|
||||
/// Unix timestamp in fractional seconds. Shared by the queue, chat store and
|
||||
/// link cache (previously four private copies).
|
||||
pub fn now_f64() -> f64 {
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_secs_f64())
|
||||
.unwrap_or(0.0)
|
||||
}
|
||||
|
||||
/// Unix timestamp in whole seconds. Same clock as [`now_f64`], for fields
|
||||
/// that store integer seconds (chat-state expiry, edit prompts).
|
||||
pub fn unix_now() -> i64 {
|
||||
now_f64() as i64
|
||||
}
|
||||
|
||||
@@ -1,872 +0,0 @@
|
||||
use crate::config::Config;
|
||||
use crate::link_cache::{CachedMediaKind, CachedPost, LinkCache};
|
||||
use crate::queue::PersistentTaskQueue;
|
||||
use crate::send::{self, MediaItemPayload, Task};
|
||||
use crate::state::{ChatData, ChatStore, unix_now};
|
||||
use std::collections::HashSet;
|
||||
use std::sync::LazyLock;
|
||||
use teloxide::RequestError;
|
||||
use teloxide::prelude::*;
|
||||
use teloxide::types::{
|
||||
CallbackQuery, ChatAction, ChatId, ChatKind, InlineQuery, InlineQueryResult,
|
||||
InlineQueryResultMpeg4Gif, InlineQueryResultPhoto, InlineQueryResultVideo, Message,
|
||||
MessageEntityKind, MessageId, ParseMode, Recipient, ReplyParameters,
|
||||
};
|
||||
use teloxide::utils::command::BotCommands;
|
||||
use tokio::sync::Semaphore;
|
||||
use x_media::media::Media;
|
||||
|
||||
pub static CHAT_STORE: LazyLock<ChatStore> =
|
||||
LazyLock::new(|| ChatStore::open("data/task_queue.db").expect("failed to open chat store"));
|
||||
pub static TASK_QUEUE: LazyLock<PersistentTaskQueue> =
|
||||
LazyLock::new(|| PersistentTaskQueue::new("data/task_queue.db"));
|
||||
pub static LINK_CACHE: LazyLock<LinkCache> =
|
||||
LazyLock::new(|| LinkCache::open("data/task_queue.db"));
|
||||
pub static CONFIG: LazyLock<Config> = LazyLock::new(Config::load);
|
||||
|
||||
/// Cap on concurrent per-URL processing. teloxide dispatches updates to a
|
||||
/// per-chat worker that handles them sequentially, so a batch-forward of many
|
||||
/// messages would otherwise be processed one at a time (fetch + send each,
|
||||
/// roughly a second per message). Moving the work into spawned tasks trades
|
||||
/// per-chat reply ordering for throughput; the semaphore bounds how many run
|
||||
/// at once so a big burst cannot hammer Telegram's rate limits.
|
||||
static URL_TASKS: LazyLock<Semaphore> = LazyLock::new(|| Semaphore::new(8));
|
||||
|
||||
#[derive(BotCommands, Clone)]
|
||||
#[command(
|
||||
rename_rule = "snake_case",
|
||||
description = "Turn X/Pixiv/Bluesky links into media messages"
|
||||
)]
|
||||
enum Command {
|
||||
#[command(description = "Get started")]
|
||||
Start,
|
||||
#[command(description = "Show command help")]
|
||||
Help,
|
||||
#[command(
|
||||
description = "Set forward channel (@channel or ID)",
|
||||
parse_with = "split"
|
||||
)]
|
||||
SetForwardChannel(String),
|
||||
#[command(description = "Remove forward channel")]
|
||||
RemoveForwardChannel,
|
||||
#[command(description = "Toggle edit-before-forward")]
|
||||
EditBeforeForward,
|
||||
#[command(
|
||||
description = "Reply with [] to save as template",
|
||||
parse_with = "split"
|
||||
)]
|
||||
SetTemplate(String),
|
||||
#[command(description = "Show chat state (debug)")]
|
||||
BotDict,
|
||||
#[command(description = "Set site caption format", parse_with = "split")]
|
||||
SetFormat(String),
|
||||
#[command(
|
||||
description = "Clear link cache (admin; optional URL, else all)",
|
||||
parse_with = "split"
|
||||
)]
|
||||
ClearCache(String),
|
||||
}
|
||||
|
||||
async fn reply<T>(bot: Bot, message: Message, text: T) -> Result<Message, RequestError>
|
||||
where
|
||||
T: Into<String>,
|
||||
{
|
||||
bot.send_message(message.chat.id, text)
|
||||
.reply_parameters(ReplyParameters::new(message.id).allow_sending_without_reply())
|
||||
.await
|
||||
}
|
||||
|
||||
fn now_f64() -> f64 {
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_secs_f64())
|
||||
.unwrap_or(0.0)
|
||||
}
|
||||
|
||||
/// Extracts URL and text-link entities (text + caption), deduped in order.
|
||||
pub fn extract_urls(message: &Message) -> Vec<String> {
|
||||
let mut urls = Vec::new();
|
||||
for entity in message.parse_entities().into_iter().flatten() {
|
||||
match entity.kind() {
|
||||
MessageEntityKind::Url => urls.push(entity.text().to_string()),
|
||||
MessageEntityKind::TextLink { url } => urls.push(url.to_string()),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
for entity in message.parse_caption_entities().into_iter().flatten() {
|
||||
match entity.kind() {
|
||||
MessageEntityKind::Url => urls.push(entity.text().to_string()),
|
||||
MessageEntityKind::TextLink { url } => urls.push(url.to_string()),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
let mut seen = HashSet::new();
|
||||
urls.retain(|url| seen.insert(url.clone()));
|
||||
urls
|
||||
}
|
||||
|
||||
/// Edit-before-forward: a reply to the prompt swaps the caption of the first
|
||||
/// forwarded message. Returns true when the message was consumed as an edit.
|
||||
async fn edit_message_handler(bot: &Bot, message: &Message) -> bool {
|
||||
let Some(reply) = message.reply_to_message() else {
|
||||
return false;
|
||||
};
|
||||
let chat_id = message.chat.id.0;
|
||||
let Some(text) = message.text() else {
|
||||
return false;
|
||||
};
|
||||
let chat_data = CHAT_STORE.get(chat_id).await;
|
||||
let Some(edit) = chat_data.edit_message.get(&(reply.id.0 as i64)) else {
|
||||
return false;
|
||||
};
|
||||
let Some(first_forward_id) = edit.forward_message_ids.first() else {
|
||||
return false;
|
||||
};
|
||||
let link = format!(
|
||||
"<a href=\"{0}\">{1}</a>",
|
||||
edit.url,
|
||||
html_escape::encode_text(text)
|
||||
);
|
||||
let new_text = if edit.template.is_empty() {
|
||||
link
|
||||
} else {
|
||||
chat_data
|
||||
.template
|
||||
.get(&edit.template)
|
||||
.map(|template| template.replace("[]", &link))
|
||||
.unwrap_or(link)
|
||||
};
|
||||
let result = bot
|
||||
.edit_message_caption(ChatId(chat_id), MessageId(*first_forward_id as i32))
|
||||
.caption(new_text)
|
||||
.parse_mode(ParseMode::Html)
|
||||
.await;
|
||||
match result {
|
||||
Ok(_) => log::info!(
|
||||
"edit-before-forward: caption swapped on message {first_forward_id} for prompt {}",
|
||||
reply.id.0
|
||||
),
|
||||
Err(e) => log::error!("edit_message_caption failed: {e}"),
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
enum SetForwardChannelError {
|
||||
EmptyParameter,
|
||||
NotChannel,
|
||||
NotAdmin,
|
||||
NotBotAdmin(RequestError),
|
||||
NotBotCanPost,
|
||||
}
|
||||
|
||||
async fn set_forward_channel_handler(
|
||||
bot: &Bot,
|
||||
message: &Message,
|
||||
channel: String,
|
||||
) -> Result<i64, SetForwardChannelError> {
|
||||
if channel.is_empty() {
|
||||
return Err(SetForwardChannelError::EmptyParameter);
|
||||
}
|
||||
let channel = match channel.parse::<i64>() {
|
||||
Ok(id) => Recipient::Id(ChatId(id)),
|
||||
Err(_) => Recipient::ChannelUsername(channel),
|
||||
};
|
||||
if let Some(from) = &message.from {
|
||||
log::info!(
|
||||
"Set forward channel for {} ({}) to {}",
|
||||
from.full_name(),
|
||||
message.chat.id,
|
||||
channel
|
||||
);
|
||||
}
|
||||
let chat = match bot.get_chat(channel.clone()).await {
|
||||
Err(e) => {
|
||||
log::error!("Failed to get channel {}: {}", channel, e);
|
||||
return Err(SetForwardChannelError::NotBotAdmin(e));
|
||||
}
|
||||
Ok(chat) => chat,
|
||||
};
|
||||
if !chat.is_channel() {
|
||||
return Err(SetForwardChannelError::NotChannel);
|
||||
}
|
||||
let channel_id = chat.id.0;
|
||||
match bot.get_chat_administrators(channel.clone()).await {
|
||||
Err(e) => {
|
||||
log::error!("Failed to get channel administrators {}: {}", channel, e);
|
||||
return Err(SetForwardChannelError::NotBotAdmin(e));
|
||||
}
|
||||
Ok(admins) => {
|
||||
if !admins.iter().any(|admin| admin.user.id == message.chat.id) {
|
||||
return Err(SetForwardChannelError::NotAdmin);
|
||||
}
|
||||
let bot_id = bot.get_me().await.expect("Failed get bot id").user.id;
|
||||
if let Some(bot_admin) = admins.iter().find(|admin| admin.user.id == bot_id)
|
||||
&& !bot_admin.can_post_messages()
|
||||
{
|
||||
return Err(SetForwardChannelError::NotBotCanPost);
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(channel_id)
|
||||
}
|
||||
|
||||
async fn execute_command(
|
||||
bot: &Bot,
|
||||
message: &Message,
|
||||
command: Command,
|
||||
) -> Result<(), RequestError> {
|
||||
match command {
|
||||
Command::Start => {
|
||||
bot.send_message(message.chat.id, "Hello!").await?;
|
||||
}
|
||||
Command::Help => {
|
||||
bot.send_message(message.chat.id, Command::descriptions().to_string())
|
||||
.await?;
|
||||
}
|
||||
Command::SetForwardChannel(channel) => {
|
||||
let result = match set_forward_channel_handler(bot, message, channel).await {
|
||||
Ok(channel_id) => {
|
||||
let mut chat_data = CHAT_STORE.get(message.chat.id.0).await;
|
||||
chat_data.forward_channel_id = Some(channel_id);
|
||||
CHAT_STORE.set(message.chat.id.0, &chat_data).await;
|
||||
"Add successfully.".to_string()
|
||||
}
|
||||
Err(SetForwardChannelError::EmptyParameter) => {
|
||||
"Receive empty parameter.\nYou should enter a channel id or username"
|
||||
.to_string()
|
||||
}
|
||||
Err(SetForwardChannelError::NotChannel) => {
|
||||
"Given id / username is not a channel".to_string()
|
||||
}
|
||||
Err(SetForwardChannelError::NotAdmin) => {
|
||||
"You are not an administrator of the channel".to_string()
|
||||
}
|
||||
Err(SetForwardChannelError::NotBotAdmin(e)) => {
|
||||
e.to_string() + "\nPlease add the bot as an admin to the channel"
|
||||
}
|
||||
Err(SetForwardChannelError::NotBotCanPost) => {
|
||||
"Bot can't post messages to the channel".to_string()
|
||||
}
|
||||
};
|
||||
reply(bot.clone(), message.clone(), result).await?;
|
||||
}
|
||||
Command::RemoveForwardChannel => {
|
||||
let chat_id = message.chat.id.0;
|
||||
let mut chat_data = CHAT_STORE.get(chat_id).await;
|
||||
let text = if chat_data.forward_channel_id.is_some() {
|
||||
chat_data.forward_channel_id = None;
|
||||
CHAT_STORE.set(chat_id, &chat_data).await;
|
||||
"Remove successfully.".to_string()
|
||||
} else {
|
||||
"No channel to remove.".to_string()
|
||||
};
|
||||
reply(bot.clone(), message.clone(), text).await?;
|
||||
}
|
||||
Command::EditBeforeForward => {
|
||||
let chat_id = message.chat.id.0;
|
||||
let mut chat_data = CHAT_STORE.get(chat_id).await;
|
||||
let text = if chat_data.forward_channel_id.is_none() {
|
||||
"Please enable forward channel first.".to_string()
|
||||
} else if chat_data.edit_before_forward {
|
||||
chat_data.edit_before_forward = false;
|
||||
chat_data.edit_message.clear();
|
||||
CHAT_STORE.set(chat_id, &chat_data).await;
|
||||
"Disable edit before forward.".to_string()
|
||||
} else {
|
||||
chat_data.edit_before_forward = true;
|
||||
CHAT_STORE.set(chat_id, &chat_data).await;
|
||||
"Enable edit before forward.".to_string()
|
||||
};
|
||||
reply(bot.clone(), message.clone(), text).await?;
|
||||
}
|
||||
Command::SetTemplate(name) => {
|
||||
let chat_id = message.chat.id.0;
|
||||
let text = match message.reply_to_message() {
|
||||
None => "Please reply to a message to set as template.".to_string(),
|
||||
Some(reply) => {
|
||||
let reply_text = reply.text().unwrap_or_default();
|
||||
if !reply_text.contains("[]") {
|
||||
"Please reply to a message with [] to set as template.".to_string()
|
||||
} else if name.is_empty() {
|
||||
"Please provide a name for the template.".to_string()
|
||||
} else {
|
||||
let mut chat_data = CHAT_STORE.get(chat_id).await;
|
||||
chat_data
|
||||
.template
|
||||
.insert(name, html_escape::encode_text(reply_text).into_owned());
|
||||
CHAT_STORE.set(chat_id, &chat_data).await;
|
||||
"Template set.".to_string()
|
||||
}
|
||||
}
|
||||
};
|
||||
reply(bot.clone(), message.clone(), text).await?;
|
||||
}
|
||||
Command::BotDict => {
|
||||
let chat_data = CHAT_STORE.get(message.chat.id.0).await;
|
||||
let debug = format!("{chat_data:?}");
|
||||
let text = html_escape::encode_text(&debug).into_owned();
|
||||
reply(bot.clone(), message.clone(), text).await?;
|
||||
}
|
||||
Command::SetFormat(arg) => {
|
||||
let chat_id = message.chat.id.0;
|
||||
let (site, format) = match arg.split_once(char::is_whitespace) {
|
||||
Some((site, format)) if !format.trim().is_empty() => {
|
||||
(site.trim(), format.trim().to_string())
|
||||
}
|
||||
_ => {
|
||||
reply(
|
||||
bot.clone(),
|
||||
message.clone(),
|
||||
"Usage: /set_format <site> <format>",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
if !["twitter", "bsky", "pixiv"].contains(&site) {
|
||||
reply(
|
||||
bot.clone(),
|
||||
message.clone(),
|
||||
"Unknown site. Use twitter, bsky or pixiv.",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
let mut chat_data = CHAT_STORE.get(chat_id).await;
|
||||
chat_data.message_format.insert(site.to_string(), format);
|
||||
CHAT_STORE.set(chat_id, &chat_data).await;
|
||||
reply(bot.clone(), message.clone(), "Format set.").await?;
|
||||
}
|
||||
Command::ClearCache(arg) => {
|
||||
let sender_id = message
|
||||
.from
|
||||
.as_ref()
|
||||
.map(|user| user.id.0 as i64)
|
||||
.unwrap_or(-1);
|
||||
if !CONFIG.admin_ids.contains(&sender_id) {
|
||||
reply(bot.clone(), message.clone(), "Admin only.").await?;
|
||||
return Ok(());
|
||||
}
|
||||
let arg = arg.trim();
|
||||
if arg.is_empty() {
|
||||
let removed = LINK_CACHE.clear(None).await;
|
||||
log::info!("cache cleared by {sender_id}: {removed} entries");
|
||||
reply(
|
||||
bot.clone(),
|
||||
message.clone(),
|
||||
format!("Cleared {removed} cached entr{}.", plural(removed)),
|
||||
)
|
||||
.await?;
|
||||
} else {
|
||||
let key = match x_media::site::cache_key(arg) {
|
||||
Some(key) => key,
|
||||
None => {
|
||||
reply(
|
||||
bot.clone(),
|
||||
message.clone(),
|
||||
"Unrecognized link. Use a twitter/x, pixiv or bsky post URL.",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
let removed = LINK_CACHE.clear(Some(&key)).await;
|
||||
log::info!("cache entry cleared by {sender_id}: {key} ({removed} rows)");
|
||||
reply(
|
||||
bot.clone(),
|
||||
message.clone(),
|
||||
format!(
|
||||
"Cleared cache for {arg} ({} entr{}).",
|
||||
removed,
|
||||
plural(removed)
|
||||
),
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// `""` for one, `"ies"` for anything else — "1 entry" / "2 entries".
|
||||
fn plural(n: usize) -> &'static str {
|
||||
if n == 1 { "" } else { "ies" }
|
||||
}
|
||||
|
||||
/// For locally produced media (encoded ugoira MP4) the thumbnail URL is a
|
||||
/// hotlink-protected remote URL Telegram may not fetch; let Telegram generate
|
||||
/// its own thumbnail instead.
|
||||
fn thumbnail_for(media: &Media) -> Option<String> {
|
||||
let url = media.url();
|
||||
if url.starts_with("http://") || url.starts_with("https://") {
|
||||
media.thumbnail_url().map(str::to_string)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn media_to_payload(media: &Media, sensitive: bool) -> MediaItemPayload {
|
||||
let fallback_url = media.smaller_url().map(str::to_string);
|
||||
match media {
|
||||
// A gif inside a group becomes a video item; a lone gif takes the
|
||||
// animation path (see url_media).
|
||||
Media::Illustration { .. } => MediaItemPayload::Photo {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
Media::Video { .. } => MediaItemPayload::Video {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
thumbnail: thumbnail_for(media),
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
Media::Animated { .. } => MediaItemPayload::Video {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
thumbnail: thumbnail_for(media),
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
async fn enqueue_retry(task: Task, delay_seconds: f64) {
|
||||
let payload = serde_json::to_value(task).expect("task serializes");
|
||||
let run_after = now_f64() + delay_seconds;
|
||||
if let Err(e) = TASK_QUEUE.enqueue(payload, run_after).await {
|
||||
log::error!("failed to enqueue retry: {e}");
|
||||
}
|
||||
}
|
||||
|
||||
/// Sends a task and handles the outcome: post-send actions on success, retry
|
||||
/// enqueue on retryable failure, reply + link-cache invalidation on
|
||||
/// permanent failure (a stale cached file id must not repeat forever).
|
||||
async fn dispatch_send(bot: Bot, message: &Message, task: &Task, url: &str) {
|
||||
let result = match task {
|
||||
Task::SendAnimation { .. } => send::send_animation(&bot, task).await,
|
||||
Task::SendMediaSequence { .. } => send::send_media_sequence(&bot, task).await,
|
||||
Task::ForwardMessages { .. } => unreachable!(),
|
||||
};
|
||||
match result {
|
||||
Ok(message_ids) => {
|
||||
log::info!("sent {} message(s) for {url}", message_ids.len());
|
||||
send::post_send_actions(&bot, task, message_ids).await;
|
||||
}
|
||||
Err(send::SendError::Retryable {
|
||||
delay_seconds,
|
||||
task,
|
||||
}) => {
|
||||
log::info!("send for {url} failed, queued for retry in {delay_seconds:.1}s");
|
||||
enqueue_retry(task, delay_seconds).await;
|
||||
let _ = reply(bot, message.clone(), "Send failed. Task queued for retry.").await;
|
||||
}
|
||||
Err(send::SendError::Permanent {
|
||||
message: err_message,
|
||||
task,
|
||||
}) => {
|
||||
send::invalidate_cache(&task).await;
|
||||
log::error!("send for {url} failed permanently: {err_message}");
|
||||
let _ = reply(bot, message.clone(), format!("Send failed: {err_message}")).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Builds the send task from ready-made items, sharing the payload shape
|
||||
/// between the fresh-fetch and link-cache paths.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn build_send_task(
|
||||
chat_data: &ChatData,
|
||||
message: &Message,
|
||||
source_url: String,
|
||||
caption: String,
|
||||
items: Vec<MediaItemPayload>,
|
||||
cache_data: Option<CachedPost>,
|
||||
) -> Task {
|
||||
let chat_id = message.chat.id.0;
|
||||
if items.len() == 1 && matches!(items[0], MediaItemPayload::Animation { .. }) {
|
||||
Task::SendAnimation {
|
||||
chat_id,
|
||||
reply_to_message_id: message.id.0 as i64,
|
||||
caption,
|
||||
animation: items.into_iter().next().unwrap(),
|
||||
source_url,
|
||||
edit_before_forward: chat_data.edit_before_forward,
|
||||
forward_channel_id: chat_data.forward_channel_id,
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(message.id.0 as i64),
|
||||
cache_data,
|
||||
}
|
||||
} else {
|
||||
Task::SendMediaSequence {
|
||||
chat_id,
|
||||
reply_to_message_id: message.id.0 as i64,
|
||||
caption,
|
||||
media_batches: send::chunk_media_items(items),
|
||||
batch_index: 0,
|
||||
sent_message_ids: vec![],
|
||||
source_url,
|
||||
edit_before_forward: chat_data.edit_before_forward,
|
||||
forward_channel_id: chat_data.forward_channel_id,
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(message.id.0 as i64),
|
||||
cache_data,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn url_media(bot: Bot, message: &Message, url: &str) {
|
||||
let chat_id = message.chat.id.0;
|
||||
if let Err(e) = bot
|
||||
.send_chat_action(ChatId(chat_id), ChatAction::Typing)
|
||||
.await
|
||||
{
|
||||
log::error!("send_chat_action failed: {e}");
|
||||
}
|
||||
|
||||
// Link cache: a post sent before is re-sent from Telegram file ids —
|
||||
// no source-site request, no download, no upload. Keyed by the
|
||||
// normalized post id so x.com / fxtwitter / /photo/N variants collide.
|
||||
if let Some(key) = x_media::site::cache_key(url)
|
||||
&& let Some(cached) = LINK_CACHE.get(&key, CONFIG.link_cache_ttl).await
|
||||
{
|
||||
log::info!("link cache hit for {url}");
|
||||
let chat_data = CHAT_STORE.get(chat_id).await;
|
||||
let site = key.split(':').next().unwrap_or("unknown");
|
||||
let format = chat_data
|
||||
.message_format
|
||||
.get(site)
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let caption = if format.is_empty() {
|
||||
cached.caption.clone()
|
||||
} else {
|
||||
x_media::site::caption_from_fields(
|
||||
&format,
|
||||
"",
|
||||
&cached.url,
|
||||
&cached.author,
|
||||
&cached.author_url,
|
||||
&cached.title,
|
||||
&cached.tags,
|
||||
)
|
||||
};
|
||||
let items: Vec<MediaItemPayload> = cached
|
||||
.media
|
||||
.iter()
|
||||
.map(|m| match m.kind {
|
||||
CachedMediaKind::Photo => MediaItemPayload::Photo {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
fallback_url: None,
|
||||
file_id: true,
|
||||
},
|
||||
CachedMediaKind::Video => MediaItemPayload::Video {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
thumbnail: None,
|
||||
fallback_url: None,
|
||||
file_id: true,
|
||||
},
|
||||
CachedMediaKind::Animation => MediaItemPayload::Animation {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
file_id: true,
|
||||
},
|
||||
})
|
||||
.collect();
|
||||
let task = build_send_task(
|
||||
&chat_data,
|
||||
message,
|
||||
cached.url.clone(),
|
||||
caption,
|
||||
items,
|
||||
Some(cached),
|
||||
);
|
||||
dispatch_send(bot, message, &task, url).await;
|
||||
return;
|
||||
}
|
||||
|
||||
log::info!("fetching {url}");
|
||||
match x_media::site::fetch(url).await {
|
||||
// Unsupported links are ignored silently (Python parity).
|
||||
Ok(None) => {
|
||||
log::info!("no site pattern matches {url}; ignoring");
|
||||
}
|
||||
// Retries exhausted: notify the user (Rust-only requirement 3).
|
||||
Err(e) => {
|
||||
log::error!("fetch {url}: {e}");
|
||||
let _ = reply(
|
||||
bot,
|
||||
message.clone(),
|
||||
"Failed to fetch media from this link.",
|
||||
)
|
||||
.await;
|
||||
}
|
||||
Ok(Some(fetched)) => {
|
||||
if fetched.media.is_empty() {
|
||||
let _ = reply(
|
||||
bot,
|
||||
message.clone(),
|
||||
"No media found or media type is not supported.",
|
||||
)
|
||||
.await;
|
||||
return;
|
||||
}
|
||||
let chat_data = CHAT_STORE.get(chat_id).await;
|
||||
// Per-site caption format override (empty -> built-in caption).
|
||||
let format = chat_data
|
||||
.message_format
|
||||
.get(fetched.site_name())
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let caption = fetched.caption_with(&format);
|
||||
// Raw render data for the link cache; the send fills in the
|
||||
// Telegram file ids and persists the entry.
|
||||
let cache_data = fetched
|
||||
.render_fields()
|
||||
.map(|(author, author_url, title, tags)| CachedPost {
|
||||
url: fetched.source_url.clone(),
|
||||
caption: fetched.caption.clone(),
|
||||
title: title.to_string(),
|
||||
author: author.to_string(),
|
||||
author_url: author_url.to_string(),
|
||||
tags: tags.to_string(),
|
||||
sensitive: fetched.sensitive,
|
||||
media: vec![],
|
||||
});
|
||||
let items: Vec<MediaItemPayload> = fetched
|
||||
.media
|
||||
.iter()
|
||||
.map(|media| media_to_payload(media, fetched.sensitive))
|
||||
.collect();
|
||||
let task = build_send_task(
|
||||
&chat_data,
|
||||
message,
|
||||
fetched.source_url.clone(),
|
||||
caption,
|
||||
items,
|
||||
cache_data,
|
||||
);
|
||||
dispatch_send(bot, message, &task, url).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn message_handler(bot: Bot, message: Message) -> Result<(), RequestError> {
|
||||
let is_private = matches!(message.chat.kind, ChatKind::Private(_));
|
||||
let sender = message
|
||||
.from
|
||||
.as_ref()
|
||||
.map(|from| from.full_name())
|
||||
.unwrap_or_else(|| "unknown".to_string());
|
||||
let text_preview = message
|
||||
.text()
|
||||
.map(|t| if t.len() > 120 { &t[..120] } else { t })
|
||||
.unwrap_or("<no text>");
|
||||
log::info!(
|
||||
"message from {sender} in {} (private={is_private}): {text_preview}",
|
||||
message.chat.id
|
||||
);
|
||||
// URL/edit flows only run in private chats; commands run in any chat.
|
||||
if is_private && edit_message_handler(&bot, &message).await {
|
||||
return respond(());
|
||||
}
|
||||
if let Some(text) = message.text()
|
||||
&& let Ok(command) = Command::parse(text, "")
|
||||
{
|
||||
log::info!("command from {}: {text_preview}", message.chat.id);
|
||||
execute_command(&bot, &message, command).await?;
|
||||
return respond(());
|
||||
}
|
||||
if is_private {
|
||||
let urls = extract_urls(&message);
|
||||
if !urls.is_empty() {
|
||||
log::info!("extracted {} URL(s): {urls:?}", urls.len());
|
||||
}
|
||||
for url in urls {
|
||||
let bot = bot.clone();
|
||||
let message = message.clone();
|
||||
tokio::spawn(async move {
|
||||
// Held for the whole task; the semaphore is never closed.
|
||||
let _permit = URL_TASKS.acquire().await.expect("URL semaphore closed");
|
||||
url_media(bot, &message, &url).await;
|
||||
});
|
||||
}
|
||||
}
|
||||
respond(())
|
||||
}
|
||||
|
||||
pub async fn inline_query_handler(bot: Bot, query: InlineQuery) -> Result<(), RequestError> {
|
||||
if query.query.is_empty() {
|
||||
return respond(());
|
||||
}
|
||||
log::info!("inline query: {}", query.query);
|
||||
match x_media::site::fetch(&query.query).await {
|
||||
Ok(Some(fetched)) => {
|
||||
let mut results: Vec<InlineQueryResult> = Vec::new();
|
||||
for (i, media) in fetched.media.iter().enumerate() {
|
||||
let id = format!("{i}");
|
||||
let Some(url) = url::Url::parse(media.url()).ok() else {
|
||||
continue;
|
||||
};
|
||||
let thumbnail = media
|
||||
.thumbnail_url()
|
||||
.and_then(|t| url::Url::parse(t).ok())
|
||||
.unwrap_or_else(|| url.clone());
|
||||
let caption = fetched.caption.clone();
|
||||
let result = match media {
|
||||
Media::Illustration { .. } => {
|
||||
// Inline photo results have their own (smaller) size
|
||||
// cap; use the reduced variant when one exists.
|
||||
let photo_url = media
|
||||
.smaller_url()
|
||||
.and_then(|u| url::Url::parse(u).ok())
|
||||
.unwrap_or_else(|| url.clone());
|
||||
InlineQueryResult::Photo(
|
||||
InlineQueryResultPhoto::new(id, photo_url, thumbnail)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html),
|
||||
)
|
||||
}
|
||||
Media::Video { .. } => InlineQueryResult::Video(
|
||||
InlineQueryResultVideo::new(
|
||||
id,
|
||||
url,
|
||||
"video/mp4".parse().expect("valid mime"),
|
||||
thumbnail,
|
||||
fetched.title.clone(),
|
||||
)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html),
|
||||
),
|
||||
Media::Animated { .. } => InlineQueryResult::Mpeg4Gif(
|
||||
InlineQueryResultMpeg4Gif::new(id, url, thumbnail)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html),
|
||||
),
|
||||
};
|
||||
results.push(result);
|
||||
}
|
||||
if !results.is_empty() {
|
||||
bot.answer_inline_query(query.id, results).await?;
|
||||
}
|
||||
}
|
||||
Ok(None) => {}
|
||||
Err(e) => log::error!("inline fetch {}: {e}", query.query),
|
||||
}
|
||||
respond(())
|
||||
}
|
||||
|
||||
pub async fn callback_query_handler(bot: Bot, query: CallbackQuery) -> Result<(), RequestError> {
|
||||
let callback_query_id = query.id;
|
||||
let data = query.data.clone();
|
||||
let Some(message) = &query.message else {
|
||||
return respond(());
|
||||
};
|
||||
let chat_id = message.chat().id.0;
|
||||
let prompt_message_id = message.id().0 as i64;
|
||||
let ttl_secs = CONFIG.edit_message_ttl.as_secs() as i64;
|
||||
let mut chat_data = CHAT_STORE.get(chat_id).await;
|
||||
let edit = chat_data.edit_message.get(&prompt_message_id).cloned();
|
||||
let Some(edit) = edit else {
|
||||
log::info!(
|
||||
"callback from {}: no edit record for prompt {prompt_message_id}",
|
||||
chat_id
|
||||
);
|
||||
bot.answer_callback_query(callback_query_id)
|
||||
.text("Expired")
|
||||
.await?;
|
||||
return respond(());
|
||||
};
|
||||
// Lazy expiry: a stale record (past the TTL, not yet swept) is dropped.
|
||||
if edit.created_at + ttl_secs <= unix_now() {
|
||||
chat_data.edit_message.remove(&prompt_message_id);
|
||||
CHAT_STORE.set(chat_id, &chat_data).await;
|
||||
bot.answer_callback_query(callback_query_id)
|
||||
.text("Expired")
|
||||
.await?;
|
||||
return respond(());
|
||||
}
|
||||
|
||||
let Some(data) = data else {
|
||||
return respond(());
|
||||
};
|
||||
log::info!(
|
||||
"callback from {} on prompt {prompt_message_id}: {data}",
|
||||
chat_id
|
||||
);
|
||||
if data == "forward" {
|
||||
match chat_data.forward_channel_id {
|
||||
Some(channel_id) => {
|
||||
let forward_task = Task::ForwardMessages {
|
||||
from_chat_id: edit.chat_id,
|
||||
to_chat_id: channel_id,
|
||||
message_ids: edit.forward_message_ids.clone(),
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(prompt_message_id),
|
||||
};
|
||||
match send::forward_messages(&bot, &forward_task).await {
|
||||
Ok(()) => {
|
||||
log::info!(
|
||||
"forwarded {} message(s) to channel {channel_id}",
|
||||
edit.forward_message_ids.len()
|
||||
);
|
||||
bot.answer_callback_query(callback_query_id)
|
||||
.text("✅ Forwarded")
|
||||
.await?;
|
||||
let _ = bot
|
||||
.delete_message(ChatId(chat_id), MessageId(prompt_message_id as i32))
|
||||
.await;
|
||||
chat_data.edit_message.remove(&prompt_message_id);
|
||||
CHAT_STORE.set(chat_id, &chat_data).await;
|
||||
}
|
||||
Err(send::SendError::Retryable {
|
||||
delay_seconds,
|
||||
task,
|
||||
}) => {
|
||||
log::info!("forward queued for retry in {delay_seconds:.1}s");
|
||||
enqueue_retry(task, delay_seconds).await;
|
||||
bot.answer_callback_query(callback_query_id)
|
||||
.text("Forward queued for retry.")
|
||||
.await?;
|
||||
}
|
||||
Err(send::SendError::Permanent { message, .. }) => {
|
||||
log::error!("forward failed permanently: {message}");
|
||||
bot.answer_callback_query(callback_query_id)
|
||||
.text(format!("Forward failed: {message}"))
|
||||
.await?;
|
||||
}
|
||||
}
|
||||
}
|
||||
None => {
|
||||
log::info!("forward callback without a forward channel set");
|
||||
bot.answer_callback_query(callback_query_id)
|
||||
.text("No forward channel set.")
|
||||
.await?;
|
||||
}
|
||||
}
|
||||
return respond(());
|
||||
}
|
||||
if let Some(name) = data.strip_prefix("template|") {
|
||||
if let Some(template_html) = chat_data.template.get(name).cloned()
|
||||
&& let Some(first_forward_id) = edit.forward_message_ids.first().copied()
|
||||
{
|
||||
// Raw template including the [] placeholder (Python parity).
|
||||
let _ = bot
|
||||
.edit_message_caption(ChatId(chat_id), MessageId(first_forward_id as i32))
|
||||
.caption(template_html)
|
||||
.parse_mode(ParseMode::Html)
|
||||
.await;
|
||||
if let Some(entry) = chat_data.edit_message.get_mut(&prompt_message_id) {
|
||||
entry.template = name.to_string();
|
||||
}
|
||||
CHAT_STORE.set(chat_id, &chat_data).await;
|
||||
log::info!("template '{name}' applied to prompt {prompt_message_id}");
|
||||
}
|
||||
bot.answer_callback_query(callback_query_id).await?;
|
||||
}
|
||||
respond(())
|
||||
}
|
||||
@@ -0,0 +1,321 @@
|
||||
//! Callback query handling: the edit-before-forward prompt's `"forward"` and
|
||||
//! `"template|<name>"` buttons.
|
||||
//!
|
||||
//! [`callback_query_handler`] is the dptree entry; it only pulls the plain
|
||||
//! values out of the teloxide update and hands them to [`handle_callback`],
|
||||
//! which holds the button logic and is driven directly by tests.
|
||||
|
||||
use crate::ctx::AppContext;
|
||||
use crate::db::unix_now;
|
||||
use crate::send::{self, Task};
|
||||
use teloxide::RequestError;
|
||||
use teloxide::prelude::*;
|
||||
use teloxide::types::{CallbackQuery, CallbackQueryId, MessageId};
|
||||
|
||||
/// The `"forward"` button's data.
|
||||
const FORWARD: &str = "forward";
|
||||
/// Prefix of a template button's data: `"template|<name>"`.
|
||||
const TEMPLATE_PREFIX: &str = "template|";
|
||||
|
||||
pub async fn callback_query_handler(bot: Bot, query: CallbackQuery) -> Result<(), RequestError> {
|
||||
let Some(message) = &query.message else {
|
||||
return respond(());
|
||||
};
|
||||
let Some(data) = query.data.clone() else {
|
||||
return respond(());
|
||||
};
|
||||
let ctx = AppContext::from_statics(&bot);
|
||||
handle_callback(
|
||||
&ctx,
|
||||
query.id.clone(),
|
||||
message.chat().id.0,
|
||||
message.id().0 as i64,
|
||||
&data,
|
||||
)
|
||||
.await;
|
||||
respond(())
|
||||
}
|
||||
|
||||
/// Handles one button press on the edit-before-forward prompt.
|
||||
async fn handle_callback(
|
||||
ctx: &AppContext<'_>,
|
||||
callback_query_id: CallbackQueryId,
|
||||
chat_id: i64,
|
||||
prompt_message_id: i64,
|
||||
data: &str,
|
||||
) {
|
||||
let ttl_secs = ctx.config.edit_message_ttl.as_secs() as i64;
|
||||
let chat_data = ctx.chat_store.get(chat_id).await;
|
||||
let edit = chat_data.edit_message.get(&prompt_message_id).cloned();
|
||||
let Some(edit) = edit else {
|
||||
log::debug!("callback from {chat_id}: no edit record for prompt {prompt_message_id}");
|
||||
let _ = ctx
|
||||
.sender
|
||||
.answer_callback_query(callback_query_id, Some("Expired".to_string()))
|
||||
.await;
|
||||
return;
|
||||
};
|
||||
// Lazy expiry: a stale record (past the TTL, not yet swept) is dropped.
|
||||
if edit.created_at + ttl_secs <= unix_now() {
|
||||
ctx.chat_store
|
||||
.update(chat_id, |data| {
|
||||
data.edit_message.remove(&prompt_message_id);
|
||||
})
|
||||
.await;
|
||||
let _ = ctx
|
||||
.sender
|
||||
.answer_callback_query(callback_query_id, Some("Expired".to_string()))
|
||||
.await;
|
||||
return;
|
||||
}
|
||||
|
||||
log::info!("callback from {chat_id} on prompt {prompt_message_id}: {data}");
|
||||
if data == FORWARD {
|
||||
match chat_data.forward_channel_id {
|
||||
Some(channel_id) => {
|
||||
let forward_task = Task::ForwardMessages {
|
||||
from_chat_id: edit.chat_id,
|
||||
to_chat_id: channel_id,
|
||||
message_ids: edit.forward_message_ids.clone(),
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(prompt_message_id),
|
||||
};
|
||||
let (answer, settled) = match send::forward_messages(ctx, &forward_task).await {
|
||||
Ok(()) => {
|
||||
log::info!(
|
||||
"forwarded {} message(s) to channel {channel_id}",
|
||||
edit.forward_message_ids.len()
|
||||
);
|
||||
("✅ Forwarded".to_string(), true)
|
||||
}
|
||||
Err(send::SendError::Retryable {
|
||||
delay_seconds,
|
||||
task,
|
||||
}) => {
|
||||
log::info!("forward queued for retry in {delay_seconds:.1}s");
|
||||
send::enqueue_retry(ctx.task_queue, *task, delay_seconds).await;
|
||||
("Forward queued for retry.".to_string(), false)
|
||||
}
|
||||
Err(send::SendError::Permanent { message, .. }) => {
|
||||
log::error!("forward failed permanently: {message}");
|
||||
(format!("Forward failed: {message}"), false)
|
||||
}
|
||||
};
|
||||
if settled {
|
||||
// The prompt is done: drop it and its record.
|
||||
let _ = ctx
|
||||
.sender
|
||||
.delete_message(ChatId(chat_id), MessageId(prompt_message_id as i32))
|
||||
.await;
|
||||
ctx.chat_store
|
||||
.update(chat_id, |data| {
|
||||
data.edit_message.remove(&prompt_message_id);
|
||||
})
|
||||
.await;
|
||||
}
|
||||
let _ = ctx
|
||||
.sender
|
||||
.answer_callback_query(callback_query_id, Some(answer))
|
||||
.await;
|
||||
}
|
||||
None => {
|
||||
log::debug!("forward callback without a forward channel set");
|
||||
let _ = ctx
|
||||
.sender
|
||||
.answer_callback_query(
|
||||
callback_query_id,
|
||||
Some("No forward channel set.".to_string()),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if let Some(name) = data.strip_prefix(TEMPLATE_PREFIX) {
|
||||
if let Some(template_html) = chat_data.template.get(name).cloned()
|
||||
&& let Some(first_forward_id) = edit.forward_message_ids.first().copied()
|
||||
{
|
||||
// Raw template including the [] placeholder (Python parity).
|
||||
let _ = ctx
|
||||
.sender
|
||||
.edit_message_caption(
|
||||
ChatId(chat_id),
|
||||
MessageId(first_forward_id as i32),
|
||||
template_html,
|
||||
)
|
||||
.await;
|
||||
ctx.chat_store
|
||||
.update(chat_id, |data| {
|
||||
if let Some(entry) = data.edit_message.get_mut(&prompt_message_id) {
|
||||
entry.template = name.to_string();
|
||||
}
|
||||
})
|
||||
.await;
|
||||
log::info!("template '{name}' applied to prompt {prompt_message_id}");
|
||||
}
|
||||
let _ = ctx
|
||||
.sender
|
||||
.answer_callback_query(callback_query_id, None)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::ctx::test_support::TestStores;
|
||||
use crate::media_sender::test_support::{MockSender, Outcome};
|
||||
use crate::state::EditMessage;
|
||||
use teloxide::ApiError;
|
||||
|
||||
/// The edit-before-forward prompt's message id in these tests.
|
||||
const PROMPT_ID: i64 = 7;
|
||||
/// The message the prompt refers to (the one whose caption is swapped).
|
||||
const FORWARDED_ID: i64 = 9;
|
||||
|
||||
fn api_error() -> RequestError {
|
||||
RequestError::Api(ApiError::Unknown("Bad Request: chat not found".into()))
|
||||
}
|
||||
|
||||
fn callback_id() -> CallbackQueryId {
|
||||
CallbackQueryId("cb-1".to_string())
|
||||
}
|
||||
|
||||
/// Seeds a live prompt record plus a forward channel and a template;
|
||||
/// `created_at` backdates the record for the expiry cases.
|
||||
async fn seed_prompt(ctx: &AppContext<'_>, created_at: i64) {
|
||||
ctx.chat_store
|
||||
.update(1, |data| {
|
||||
data.forward_channel_id = Some(2);
|
||||
data.template
|
||||
.insert("tpl".to_string(), "<b>[]</b>".to_string());
|
||||
data.edit_message.insert(
|
||||
PROMPT_ID,
|
||||
EditMessage {
|
||||
url: "https://x.com/u/status/1".into(),
|
||||
chat_id: 1,
|
||||
forward_message_ids: vec![FORWARDED_ID],
|
||||
template: String::new(),
|
||||
created_at,
|
||||
},
|
||||
);
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn template_button_swaps_the_caption_and_records_the_choice() {
|
||||
let sender = MockSender::scripted(vec![Outcome::EditOk], api_error);
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
seed_prompt(&ctx, crate::db::unix_now()).await;
|
||||
|
||||
handle_callback(&ctx, callback_id(), 1, PROMPT_ID, "template|tpl").await;
|
||||
|
||||
assert_eq!(
|
||||
sender.calls(),
|
||||
vec!["edit_message_caption", "answer_callback_query"]
|
||||
);
|
||||
// The raw template, including the [] the user edits into.
|
||||
assert_eq!(sender.captions(), vec!["<b>[]</b>"]);
|
||||
assert_eq!(sender.answers(), vec![None]);
|
||||
let data = ctx.chat_store.get(1).await;
|
||||
assert_eq!(data.edit_message[&PROMPT_ID].template, "tpl");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn forward_button_copies_then_clears_the_prompt() {
|
||||
let sender = MockSender::scripted(vec![Outcome::CopyOk], api_error);
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
seed_prompt(&ctx, crate::db::unix_now()).await;
|
||||
|
||||
handle_callback(&ctx, callback_id(), 1, PROMPT_ID, "forward").await;
|
||||
|
||||
assert_eq!(
|
||||
sender.calls(),
|
||||
vec!["copy_messages", "delete_message", "answer_callback_query"]
|
||||
);
|
||||
assert_eq!(sender.answers(), vec![Some("✅ Forwarded".to_string())]);
|
||||
assert!(
|
||||
ctx.chat_store.get(1).await.edit_message.is_empty(),
|
||||
"a settled prompt must drop its record"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn forward_without_a_channel_is_reported() {
|
||||
let sender = MockSender::scripted(vec![], api_error);
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
seed_prompt(&ctx, crate::db::unix_now()).await;
|
||||
ctx.chat_store
|
||||
.update(1, |data| data.forward_channel_id = None)
|
||||
.await;
|
||||
|
||||
handle_callback(&ctx, callback_id(), 1, PROMPT_ID, "forward").await;
|
||||
|
||||
assert_eq!(sender.calls(), vec!["answer_callback_query"]);
|
||||
assert_eq!(
|
||||
sender.answers(),
|
||||
vec![Some("No forward channel set.".to_string())]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn retryable_forward_is_queued_and_keeps_the_prompt() {
|
||||
use teloxide::types::Seconds;
|
||||
let sender = MockSender::scripted(vec![Outcome::CopyErr], || {
|
||||
RequestError::RetryAfter(Seconds::from_seconds(7))
|
||||
});
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
seed_prompt(&ctx, crate::db::unix_now()).await;
|
||||
|
||||
handle_callback(&ctx, callback_id(), 1, PROMPT_ID, "forward").await;
|
||||
|
||||
assert_eq!(
|
||||
sender.calls(),
|
||||
vec!["copy_messages", "answer_callback_query"]
|
||||
);
|
||||
assert_eq!(
|
||||
sender.answers(),
|
||||
vec![Some("Forward queued for retry.".to_string())]
|
||||
);
|
||||
assert_eq!(stores.queued_tasks().await, 1);
|
||||
// The prompt is not settled: the queued retry still needs the record.
|
||||
assert!(
|
||||
ctx.chat_store
|
||||
.get(1)
|
||||
.await
|
||||
.edit_message
|
||||
.contains_key(&PROMPT_ID)
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn unknown_and_expired_prompts_answer_expired() {
|
||||
let sender = MockSender::scripted(vec![], api_error);
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
|
||||
// No record at all.
|
||||
handle_callback(&ctx, callback_id(), 1, PROMPT_ID, "forward").await;
|
||||
assert_eq!(sender.answers(), vec![Some("Expired".to_string())]);
|
||||
|
||||
// A record past its TTL (nothing swept it yet) is dropped on use.
|
||||
let stale = crate::db::unix_now() - ctx.config.edit_message_ttl.as_secs() as i64 - 1;
|
||||
seed_prompt(&ctx, stale).await;
|
||||
handle_callback(&ctx, callback_id(), 1, PROMPT_ID, "forward").await;
|
||||
assert_eq!(
|
||||
sender.answers(),
|
||||
vec![Some("Expired".to_string()), Some("Expired".to_string())]
|
||||
);
|
||||
assert!(
|
||||
ctx.chat_store.get(1).await.edit_message.is_empty(),
|
||||
"the expired record must be dropped"
|
||||
);
|
||||
assert_eq!(sender.calls(), vec!["answer_callback_query"; 2]);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,660 @@
|
||||
//! Bot command parsing, the `/`-command executor and `setMyCommands`
|
||||
//! registration. URL/inline/callback flows live in their own modules.
|
||||
|
||||
use super::urls::{PostSend, url_media};
|
||||
use super::{CHAT_STORE, CONFIG, LINK_CACHE, log_key, reply, reply_html};
|
||||
use crate::ctx::AppContext;
|
||||
use teloxide::RequestError;
|
||||
use teloxide::prelude::*;
|
||||
use teloxide::types::{ChatId, Message, Recipient};
|
||||
use teloxide::utils::command::{BotCommands, ParseError};
|
||||
|
||||
#[derive(BotCommands, Clone)]
|
||||
#[command(
|
||||
rename_rule = "snake_case",
|
||||
description = "Turn X/Pixiv/Bluesky links into media messages"
|
||||
)]
|
||||
pub(crate) enum Command {
|
||||
#[command(description = "Get started")]
|
||||
Start,
|
||||
#[command(description = "Show command help")]
|
||||
Help,
|
||||
#[command(
|
||||
description = "Set forward channel (@channel or ID)",
|
||||
parse_with = "split"
|
||||
)]
|
||||
SetForwardChannel(String),
|
||||
#[command(description = "Remove forward channel")]
|
||||
RemoveForwardChannel,
|
||||
#[command(description = "Toggle edit-before-forward")]
|
||||
EditBeforeForward,
|
||||
#[command(
|
||||
description = "Reply with [] to save as template",
|
||||
parse_with = "split"
|
||||
)]
|
||||
SetTemplate(String),
|
||||
#[command(description = "Show chat state (debug; admin only)")]
|
||||
BotDict,
|
||||
#[command(description = "Set site caption format", parse_with = "split")]
|
||||
SetFormat(String),
|
||||
#[command(
|
||||
description = "Clear link cache (admin; optional URL, else all)",
|
||||
parse_with = "split"
|
||||
)]
|
||||
ClearCache(String),
|
||||
#[command(
|
||||
description = "Send a link's media (no forwarding)",
|
||||
parse_with = parse_arg_remainder
|
||||
)]
|
||||
Test(String),
|
||||
#[command(
|
||||
description = "Parse a link and report it (debug; nothing sent)",
|
||||
parse_with = parse_arg_remainder
|
||||
)]
|
||||
Debug(String),
|
||||
}
|
||||
|
||||
/// `/test` and `/debug` argument parser: the whole remainder after the command
|
||||
/// name, trimmed. The built-in `split` parser takes exactly one space-separated
|
||||
/// token and rejects the rest, so a URL followed by a trailing space (or
|
||||
/// pasted text) would silently fall through to the URL flow instead.
|
||||
fn parse_arg_remainder(s: String) -> Result<(String,), ParseError> {
|
||||
Ok((s.trim().to_string(),))
|
||||
}
|
||||
|
||||
enum SetForwardChannelError {
|
||||
EmptyParameter,
|
||||
NotChannel,
|
||||
NotAdmin,
|
||||
NotBotAdmin(RequestError),
|
||||
NotBotCanPost,
|
||||
}
|
||||
|
||||
async fn set_forward_channel_handler(
|
||||
bot: &Bot,
|
||||
message: &Message,
|
||||
channel: String,
|
||||
) -> Result<i64, SetForwardChannelError> {
|
||||
if channel.is_empty() {
|
||||
return Err(SetForwardChannelError::EmptyParameter);
|
||||
}
|
||||
let channel = match channel.parse::<i64>() {
|
||||
Ok(id) => Recipient::Id(ChatId(id)),
|
||||
Err(_) => Recipient::ChannelUsername(channel),
|
||||
};
|
||||
if let Some(from) = &message.from {
|
||||
log::info!(
|
||||
"Set forward channel for {} ({}) to {}",
|
||||
from.full_name(),
|
||||
message.chat.id,
|
||||
channel
|
||||
);
|
||||
}
|
||||
let chat = match bot.get_chat(channel.clone()).await {
|
||||
Err(e) => {
|
||||
log::error!("Failed to get channel {}: {}", channel, e);
|
||||
return Err(SetForwardChannelError::NotBotAdmin(e));
|
||||
}
|
||||
Ok(chat) => chat,
|
||||
};
|
||||
if !chat.is_channel() {
|
||||
return Err(SetForwardChannelError::NotChannel);
|
||||
}
|
||||
let channel_id = chat.id.0;
|
||||
// The sender must be a channel administrator. Compare against the
|
||||
// sender's user id, NOT the chat id (they only coincide in private
|
||||
// chats, so the old check broke group usage).
|
||||
let Some(sender) = message.from.as_ref() else {
|
||||
return Err(SetForwardChannelError::NotAdmin);
|
||||
};
|
||||
match bot.get_chat_administrators(channel.clone()).await {
|
||||
Err(e) => {
|
||||
log::error!("Failed to get channel administrators {}: {}", channel, e);
|
||||
return Err(SetForwardChannelError::NotBotAdmin(e));
|
||||
}
|
||||
Ok(admins) => {
|
||||
if !admins.iter().any(|admin| admin.user.id == sender.id) {
|
||||
return Err(SetForwardChannelError::NotAdmin);
|
||||
}
|
||||
// The bot itself must be an admin that can post; a missing
|
||||
// bot entry must not pass silently (copy would fail later).
|
||||
let bot_id = match bot.get_me().await {
|
||||
Ok(me) => me.user.id,
|
||||
Err(e) => return Err(SetForwardChannelError::NotBotAdmin(e)),
|
||||
};
|
||||
let bot_ok = admins
|
||||
.iter()
|
||||
.any(|admin| admin.user.id == bot_id && admin.can_post_messages());
|
||||
if !bot_ok {
|
||||
return Err(SetForwardChannelError::NotBotCanPost);
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(channel_id)
|
||||
}
|
||||
|
||||
pub(crate) async fn execute_command(
|
||||
bot: &Bot,
|
||||
message: &Message,
|
||||
command: Command,
|
||||
) -> Result<(), RequestError> {
|
||||
match command {
|
||||
Command::Start => {
|
||||
bot.send_message(message.chat.id, "Hello!").await?;
|
||||
}
|
||||
Command::Help => {
|
||||
bot.send_message(message.chat.id, Command::descriptions().to_string())
|
||||
.await?;
|
||||
}
|
||||
Command::SetForwardChannel(channel) => {
|
||||
let result = match set_forward_channel_handler(bot, message, channel).await {
|
||||
Ok(channel_id) => {
|
||||
CHAT_STORE
|
||||
.update(message.chat.id.0, |data| {
|
||||
data.forward_channel_id = Some(channel_id);
|
||||
})
|
||||
.await;
|
||||
"Add successfully.".to_string()
|
||||
}
|
||||
Err(SetForwardChannelError::EmptyParameter) => {
|
||||
"Receive empty parameter.\nYou should enter a channel id or username"
|
||||
.to_string()
|
||||
}
|
||||
Err(SetForwardChannelError::NotChannel) => {
|
||||
"Given id / username is not a channel".to_string()
|
||||
}
|
||||
Err(SetForwardChannelError::NotAdmin) => {
|
||||
"You are not an administrator of the channel".to_string()
|
||||
}
|
||||
Err(SetForwardChannelError::NotBotAdmin(e)) => {
|
||||
e.to_string() + "\nPlease add the bot as an admin to the channel"
|
||||
}
|
||||
Err(SetForwardChannelError::NotBotCanPost) => {
|
||||
"Bot can't post messages to the channel".to_string()
|
||||
}
|
||||
};
|
||||
reply(bot, message.chat.id.0, message.id, result).await?;
|
||||
}
|
||||
Command::RemoveForwardChannel => {
|
||||
let chat_id = message.chat.id.0;
|
||||
let text = CHAT_STORE
|
||||
.update(chat_id, |data| {
|
||||
if data.forward_channel_id.is_some() {
|
||||
data.forward_channel_id = None;
|
||||
"Remove successfully.".to_string()
|
||||
} else {
|
||||
"No channel to remove.".to_string()
|
||||
}
|
||||
})
|
||||
.await;
|
||||
reply(bot, message.chat.id.0, message.id, text).await?;
|
||||
}
|
||||
Command::EditBeforeForward => {
|
||||
let chat_id = message.chat.id.0;
|
||||
let text = CHAT_STORE
|
||||
.update(chat_id, |data| {
|
||||
if data.forward_channel_id.is_none() {
|
||||
"Please enable forward channel first.".to_string()
|
||||
} else if data.edit_before_forward {
|
||||
data.edit_before_forward = false;
|
||||
data.edit_message.clear();
|
||||
"Disable edit before forward.".to_string()
|
||||
} else {
|
||||
data.edit_before_forward = true;
|
||||
"Enable edit before forward.".to_string()
|
||||
}
|
||||
})
|
||||
.await;
|
||||
reply(bot, message.chat.id.0, message.id, text).await?;
|
||||
}
|
||||
Command::SetTemplate(name) => {
|
||||
let chat_id = message.chat.id.0;
|
||||
let text = match message.reply_to_message() {
|
||||
None => "Please reply to a message to set as template.".to_string(),
|
||||
Some(reply) => {
|
||||
let reply_text = reply.text().unwrap_or_default();
|
||||
if !reply_text.contains("[]") {
|
||||
"Please reply to a message with [] to set as template.".to_string()
|
||||
} else if name.is_empty() {
|
||||
"Please provide a name for the template.".to_string()
|
||||
} else {
|
||||
CHAT_STORE
|
||||
.update(chat_id, |data| {
|
||||
data.template.insert(
|
||||
name,
|
||||
html_escape::encode_text(reply_text).into_owned(),
|
||||
);
|
||||
})
|
||||
.await;
|
||||
"Template set.".to_string()
|
||||
}
|
||||
}
|
||||
};
|
||||
reply(bot, message.chat.id.0, message.id, text).await?;
|
||||
}
|
||||
Command::BotDict => {
|
||||
// Debug dump of the chat's persisted state: admin only (it echoes
|
||||
// forward-channel ids and templates to whoever asks).
|
||||
let sender_id = message
|
||||
.from
|
||||
.as_ref()
|
||||
.map(|user| user.id.0 as i64)
|
||||
.unwrap_or(-1);
|
||||
if !CONFIG.admin_ids.contains(&sender_id) {
|
||||
reply(bot, message.chat.id.0, message.id, "Admin only.").await?;
|
||||
return Ok(());
|
||||
}
|
||||
let chat_data = CHAT_STORE.get(message.chat.id.0).await;
|
||||
let debug = html_escape::encode_text(&format!("{chat_data:?}")).into_owned();
|
||||
// A chat with many templates/edit records exceeds Telegram's 4096
|
||||
// char message limit; the dump is plain text (no parse mode), so a
|
||||
// plain byte-boundary cut is safe.
|
||||
let end = debug.floor_char_boundary(MAX_DEBUG_DUMP_CHARS.min(debug.len()));
|
||||
let text = if end < debug.len() {
|
||||
format!("{}…", &debug[..end])
|
||||
} else {
|
||||
debug
|
||||
};
|
||||
reply(bot, message.chat.id.0, message.id, text).await?;
|
||||
}
|
||||
Command::SetFormat(arg) => {
|
||||
let chat_id = message.chat.id.0;
|
||||
let (site, format) = match arg.split_once(char::is_whitespace) {
|
||||
Some((site, format)) if !format.trim().is_empty() => {
|
||||
(site.trim(), format.trim().to_string())
|
||||
}
|
||||
_ => {
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
"Usage: /set_format <site> <format>",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
if !x_media::site::site_ids().contains(&site) {
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
"Unknown site. Use twitter, bsky, pixiv, misskey or bilibili.",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
CHAT_STORE
|
||||
.update(chat_id, |data| {
|
||||
data.message_format.insert(site.to_string(), format);
|
||||
})
|
||||
.await;
|
||||
reply(bot, message.chat.id.0, message.id, "Format set.").await?;
|
||||
}
|
||||
Command::ClearCache(arg) => {
|
||||
let sender_id = message
|
||||
.from
|
||||
.as_ref()
|
||||
.map(|user| user.id.0 as i64)
|
||||
.unwrap_or(-1);
|
||||
if !CONFIG.admin_ids.contains(&sender_id) {
|
||||
reply(bot, message.chat.id.0, message.id, "Admin only.").await?;
|
||||
return Ok(());
|
||||
}
|
||||
let arg = arg.trim();
|
||||
if arg.is_empty() {
|
||||
let removed = LINK_CACHE.clear(None).await;
|
||||
log::info!("cache cleared by {sender_id}: {removed} entries");
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
format!("Cleared {removed} cached entr{}.", plural(removed)),
|
||||
)
|
||||
.await?;
|
||||
} else {
|
||||
let key = match x_media::site::cache_key(arg) {
|
||||
Some(key) => key,
|
||||
None => {
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
"Unrecognized link. Use a twitter/x, pixiv, bsky, misskey or bilibili post URL.",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
let removed = LINK_CACHE.clear(Some(&key)).await;
|
||||
log::info!("cache entry cleared by {sender_id}: {key} ({removed} rows)");
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
format!(
|
||||
"Cleared cache for {arg} ({} entr{}).",
|
||||
removed,
|
||||
plural(removed)
|
||||
),
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
}
|
||||
Command::Test(arg) => {
|
||||
let url = arg.trim();
|
||||
if url.is_empty() {
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
"Usage: /test <post url>",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
if x_media::site::cache_key(url).is_none() {
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
"No enabled site matches this link (twitter/x, pixiv, bsky, misskey or bilibili).",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
// The ordinary link pipeline with the chat's post-send actions
|
||||
// suppressed: the media is sent (and cached) like a normal link,
|
||||
// but nothing is forwarded to the channel and no
|
||||
// edit-before-forward prompt opens. Info level echoes the
|
||||
// normalized key (never the raw URL) per the logging convention.
|
||||
log::info!("test: sending [key={}]", log_key(url));
|
||||
let ctx = AppContext::from_statics(bot);
|
||||
url_media(
|
||||
&ctx,
|
||||
message.chat.id.0,
|
||||
message.id.0 as i64,
|
||||
url,
|
||||
PostSend::Suppressed,
|
||||
)
|
||||
.await;
|
||||
}
|
||||
Command::Debug(arg) => {
|
||||
let url = arg.trim();
|
||||
if url.is_empty() {
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
"Usage: /debug <post url>",
|
||||
)
|
||||
.await?;
|
||||
return Ok(());
|
||||
}
|
||||
// Debug tool: report the parse result only — nothing is sent,
|
||||
// cached or forwarded.
|
||||
log::info!("debug: parsing [key={}]", log_key(url));
|
||||
match x_media::site::fetch(url).await {
|
||||
Ok(None) => {
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
"No enabled site matches this link (twitter/x, pixiv, bsky, misskey or bilibili).",
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
Err(e) => {
|
||||
reply(
|
||||
bot,
|
||||
message.chat.id.0,
|
||||
message.id,
|
||||
format!("Fetch failed: {e}"),
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
Ok(Some(fetched)) => {
|
||||
let report = debug_report(
|
||||
url,
|
||||
fetched.site_name(),
|
||||
&fetched.source_url,
|
||||
&fetched.title,
|
||||
&fetched.content,
|
||||
fetched.render_fields(),
|
||||
fetched.sensitive,
|
||||
&fetched.caption,
|
||||
&fetched.media,
|
||||
);
|
||||
// HTML report: the caption renders inside a <blockquote>
|
||||
// exactly as it will appear in the sent media message.
|
||||
reply_html(bot, message.chat.id.0, message.id, report).await?;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// `"y"` for one, `"ies"` for anything else — "1 entry" / "2 entries".
|
||||
fn plural(n: usize) -> &'static str {
|
||||
if n == 1 { "y" } else { "ies" }
|
||||
}
|
||||
|
||||
/// Registers the bot's command list with Telegram so clients show it in the
|
||||
/// `/` menu (Bot API `setMyCommands`).
|
||||
pub async fn register_commands(bot: &Bot) -> Result<(), RequestError> {
|
||||
let commands = Command::bot_commands();
|
||||
bot.set_my_commands(commands.clone()).await?;
|
||||
log::info!("registered {} commands", commands.len());
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Telegram's plain-text message limit is 4096 chars; the report stays under
|
||||
/// it even for very large threads (many media lines + a long caption).
|
||||
const MAX_DEBUG_REPORT_CHARS: usize = 4000;
|
||||
|
||||
/// Cap for the `/bot_dict` debug dump: the state is echoed as one plain-text
|
||||
/// message, so it must stay under Telegram's 4096-char limit.
|
||||
const MAX_DEBUG_DUMP_CHARS: usize = 3500;
|
||||
|
||||
/// Builds the HTML report for the `/debug` command: what the parser produced
|
||||
/// for a link (site, canonical URL, title/author/tags, caption and the media
|
||||
/// list) — no media is sent and nothing is cached or forwarded. Sent with
|
||||
/// HTML parse mode: raw fields are escaped, the pre-escaped render fields are
|
||||
/// embedded as-is, and the caption is wrapped in a `<blockquote>` so it shows
|
||||
/// exactly as it will render in the sent media message. Fields are passed
|
||||
/// individually so the formatter stays a pure function testable without
|
||||
/// constructing a `Fetched` (its render fields are `pub(crate)` to the
|
||||
/// x-media crate).
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn debug_report(
|
||||
url: &str,
|
||||
site_id: &str,
|
||||
source_url: &str,
|
||||
title: &str,
|
||||
content: &str,
|
||||
render: Option<(&str, &str, &str, &str, &str)>,
|
||||
sensitive: bool,
|
||||
caption: &str,
|
||||
media: &[x_media::media::Media],
|
||||
) -> String {
|
||||
let mut lines = vec![
|
||||
format!("Parse result for {}", html_escape::encode_text(url)),
|
||||
format!("site: {site_id}"),
|
||||
format!(
|
||||
"key: {}",
|
||||
html_escape::encode_text(
|
||||
&x_media::site::cache_key(url).unwrap_or_else(|| "<unsupported>".to_string())
|
||||
)
|
||||
),
|
||||
];
|
||||
lines.push(format!(
|
||||
"source_url: {}",
|
||||
html_escape::encode_text(source_url)
|
||||
));
|
||||
lines.push(format!("title: {}", html_escape::encode_text(title)));
|
||||
lines.push(format!("content: {}", html_escape::encode_text(content)));
|
||||
if let Some((author, author_url, _title, _content, tags)) = render {
|
||||
// The render fields are already pre-escaped for HTML captions; embed
|
||||
// them as-is so the report renders them exactly like the final
|
||||
// caption. `author_url` is raw and gets escaped here.
|
||||
lines.push(format!("author: {author}"));
|
||||
lines.push(format!(
|
||||
"author_url: {}",
|
||||
html_escape::encode_text(author_url)
|
||||
));
|
||||
lines.push(format!("tags: {tags}"));
|
||||
}
|
||||
lines.push(format!("sensitive: {sensitive}"));
|
||||
// The caption is wrapped in a <blockquote> so the report (an HTML
|
||||
// message) shows it exactly as it will render in the sent media caption
|
||||
// — escaped text and links included.
|
||||
lines.push(format!(
|
||||
"caption: <blockquote>{}</blockquote>",
|
||||
x_media::site::truncate_caption(caption)
|
||||
));
|
||||
lines.push(format!("media ({}):", media.len()));
|
||||
for (i, item) in media.iter().enumerate() {
|
||||
let kind = match item {
|
||||
x_media::media::Media::Illustration { .. } => "image",
|
||||
x_media::media::Media::Video { .. } => "video",
|
||||
x_media::media::Media::Animated { .. } => "gif",
|
||||
};
|
||||
lines.push(format!(
|
||||
" {}. {kind}: {}",
|
||||
i + 1,
|
||||
html_escape::encode_text(item.url())
|
||||
));
|
||||
}
|
||||
let mut out = lines.join(
|
||||
"
|
||||
",
|
||||
);
|
||||
if out.chars().count() > MAX_DEBUG_REPORT_CHARS {
|
||||
let end = out.floor_char_boundary(MAX_DEBUG_REPORT_CHARS - 1);
|
||||
out = format!("{}…", &out[..end]);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{MAX_DEBUG_REPORT_CHARS, debug_report};
|
||||
use x_media::media::Media;
|
||||
|
||||
#[test]
|
||||
fn debug_report_renders_fields_and_media() {
|
||||
let media = vec![
|
||||
Media::Illustration {
|
||||
title: None,
|
||||
url: "https://cdn.example/1.jpg".into(),
|
||||
thumbnail_url: None,
|
||||
fallback_url: None,
|
||||
},
|
||||
Media::Video {
|
||||
title: None,
|
||||
url: "https://cdn.example/2.mp4".into(),
|
||||
thumbnail_url: "https://cdn.example/2.jpg".into(),
|
||||
},
|
||||
];
|
||||
let report = debug_report(
|
||||
"https://x.com/u/status/1",
|
||||
"twitter",
|
||||
"https://x.com/u/status/1",
|
||||
"My title",
|
||||
"My content",
|
||||
Some((
|
||||
"Author",
|
||||
"https://x.com/u",
|
||||
"My title",
|
||||
"My content",
|
||||
"tag1 tag2",
|
||||
)),
|
||||
false,
|
||||
"<a href=\"https://x.com/u\">Author</a> · My title",
|
||||
&media,
|
||||
);
|
||||
assert!(report.contains("site: twitter"), "{report}");
|
||||
assert!(report.contains("key: twitter:1"), "{report}");
|
||||
assert!(report.contains("title: My title"), "{report}");
|
||||
assert!(report.contains("content: My content"), "{report}");
|
||||
assert!(report.contains("author: Author"), "{report}");
|
||||
assert!(report.contains("author_url: https://x.com/u"), "{report}");
|
||||
assert!(report.contains("tags: tag1 tag2"), "{report}");
|
||||
assert!(report.contains("sensitive: false"), "{report}");
|
||||
assert!(report.contains("media (2):"), "{report}");
|
||||
assert!(
|
||||
report.contains("1. image: https://cdn.example/1.jpg"),
|
||||
"{report}"
|
||||
);
|
||||
assert!(
|
||||
report.contains("2. video: https://cdn.example/2.mp4"),
|
||||
"{report}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn debug_report_without_render_data_and_no_media() {
|
||||
let report = debug_report("u", "pixiv", "s", "t", "c", None, true, "p", &[]);
|
||||
assert!(!report.contains("author:"), "{report}");
|
||||
assert!(report.contains("sensitive: true"), "{report}");
|
||||
assert!(report.contains("media (0):"), "{report}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn debug_report_wraps_caption_in_blockquote() {
|
||||
// The report is an HTML message: raw fields are escaped, pre-escaped
|
||||
// render fields are embedded as-is, and the caption is wrapped in a
|
||||
// <blockquote> so it shows exactly as it will render in the sent
|
||||
// media caption (escaped text and links included).
|
||||
let report = debug_report(
|
||||
"https://x.com/u/status/1",
|
||||
"twitter",
|
||||
"https://x.com/u/status/1",
|
||||
"A & B <C>",
|
||||
"body & <more>",
|
||||
Some((
|
||||
"A & B",
|
||||
"https://x.com/u",
|
||||
"A & B <C>",
|
||||
"body & <more>",
|
||||
"#a & #b",
|
||||
)),
|
||||
false,
|
||||
"<a href=\"https://x.com/u\">A & B</a>: C <D> & E",
|
||||
&[],
|
||||
);
|
||||
// Raw fields escaped (they render back to the original text in HTML).
|
||||
assert!(report.contains("title: A & B <C>"), "{report}");
|
||||
assert!(
|
||||
report.contains("source_url: https://x.com/u/status/1"),
|
||||
"{report}"
|
||||
);
|
||||
// Pre-escaped render fields embedded as-is.
|
||||
assert!(report.contains("author: A & B"), "{report}");
|
||||
assert!(report.contains("tags: #a & #b"), "{report}");
|
||||
// Caption wrapped in a blockquote with its HTML preserved.
|
||||
assert!(
|
||||
report.contains(
|
||||
"caption: <blockquote><a href=\"https://x.com/u\">A & B</a>: C <D> & E</blockquote>"
|
||||
),
|
||||
"{report}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn debug_report_is_capped() {
|
||||
// 200 media lines ≈ 8 KB, comfortably over the cap.
|
||||
let media: Vec<Media> = (0..200)
|
||||
.map(|i| Media::Illustration {
|
||||
title: None,
|
||||
url: format!("https://cdn.example/{i}.jpg"),
|
||||
thumbnail_url: None,
|
||||
fallback_url: None,
|
||||
})
|
||||
.collect();
|
||||
let report = debug_report("u", "twitter", "s", "t", "c", None, false, "p", &media);
|
||||
assert!(report.chars().count() <= MAX_DEBUG_REPORT_CHARS, "{report}");
|
||||
assert!(report.ends_with('…'), "{report}");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,249 @@
|
||||
//! Inline query handling with a keystroke debounce: only a query stable for
|
||||
//! [`INLINE_DEBOUNCE`] triggers a fetch, and repeats are served by Telegram's
|
||||
//! inline cache instead of re-fetching.
|
||||
|
||||
use super::log_key;
|
||||
use std::collections::HashMap;
|
||||
use std::sync::LazyLock;
|
||||
use teloxide::RequestError;
|
||||
use teloxide::prelude::*;
|
||||
use teloxide::types::{
|
||||
InlineQuery, InlineQueryResult, InlineQueryResultMpeg4Gif, InlineQueryResultPhoto,
|
||||
InlineQueryResultVideo, ParseMode,
|
||||
};
|
||||
use x_media::media::Media;
|
||||
|
||||
/// Debounce window for inline queries: Telegram fires an inline query on
|
||||
/// every keystroke, and each prefix of a pasted URL (e.g. `.../status/12`,
|
||||
/// `.../status/123`, ...) already matches the site patterns. Without a
|
||||
/// debounce every keystroke triggers a fetch (3 attempts!) of a half-typed
|
||||
/// post id. Only answer once the query has been stable for this long.
|
||||
const INLINE_DEBOUNCE: std::time::Duration = std::time::Duration::from_millis(800);
|
||||
|
||||
/// Last seen inline query per user and whether it was already answered.
|
||||
/// Guards the debounce timer: a repeat of an answered query is served by
|
||||
/// Telegram's inline cache (see `cache_time`), not by another fetch. Keyed by
|
||||
/// user id — a single shared slot would let one user's typing burst (or a
|
||||
/// different user's query) cancel another user's pending answer.
|
||||
struct InlineDebounceState {
|
||||
query: String,
|
||||
answered: bool,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
struct DebounceStates(HashMap<u64, InlineDebounceState>);
|
||||
|
||||
impl DebounceStates {
|
||||
/// Records `query` as the user's newest query. Returns false when it is a
|
||||
/// repeat whose answer already went out (Telegram's inline cache serves
|
||||
/// it; re-fetching would only hit the source site again).
|
||||
fn note(&mut self, user_id: u64, query: &str) -> bool {
|
||||
if let Some(prev) = self.0.get(&user_id)
|
||||
&& prev.query == query
|
||||
&& prev.answered
|
||||
{
|
||||
return false;
|
||||
}
|
||||
self.0.insert(
|
||||
user_id,
|
||||
InlineDebounceState {
|
||||
query: query.to_string(),
|
||||
answered: false,
|
||||
},
|
||||
);
|
||||
true
|
||||
}
|
||||
|
||||
/// Claims the answer for the user's newest query; false when a newer query
|
||||
/// superseded it or the answer was already claimed.
|
||||
fn claim(&mut self, user_id: u64, query: &str) -> bool {
|
||||
let Some(state) = self.0.get_mut(&user_id) else {
|
||||
return false;
|
||||
};
|
||||
if state.query != query || state.answered {
|
||||
return false;
|
||||
}
|
||||
state.answered = true;
|
||||
true
|
||||
}
|
||||
|
||||
/// Releases a claimed-but-unsent answer so a repeat can retry the fetch.
|
||||
fn release(&mut self, user_id: u64, query: &str) {
|
||||
if let Some(state) = self.0.get_mut(&user_id)
|
||||
&& state.query == query
|
||||
{
|
||||
state.answered = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static INLINE_DEBOUNCE_STATE: LazyLock<parking_lot::Mutex<DebounceStates>> =
|
||||
LazyLock::new(|| parking_lot::Mutex::new(DebounceStates::default()));
|
||||
|
||||
pub async fn inline_query_handler(bot: Bot, query: InlineQuery) -> Result<(), RequestError> {
|
||||
if query.query.is_empty() {
|
||||
return respond(());
|
||||
}
|
||||
// Only run a fetch for something that is actually a supported post URL.
|
||||
if x_media::site::cache_key(&query.query).is_none() {
|
||||
return respond(());
|
||||
}
|
||||
// Debounce: record the query and answer only after it has been stable for
|
||||
// INLINE_DEBOUNCE (the timer below). An already-answered repeat of the
|
||||
// same query is left to Telegram's inline cache instead of re-fetching.
|
||||
let user_id = query.from.id.0;
|
||||
if !INLINE_DEBOUNCE_STATE.lock().note(user_id, &query.query) {
|
||||
return respond(());
|
||||
}
|
||||
let query_text = query.query.clone();
|
||||
tokio::spawn(async move {
|
||||
tokio::time::sleep(INLINE_DEBOUNCE).await;
|
||||
// Only the user's last query of a typing burst survives: earlier
|
||||
// timers see the query changed and give up without answering.
|
||||
if !INLINE_DEBOUNCE_STATE.lock().claim(user_id, &query_text) {
|
||||
return;
|
||||
}
|
||||
match answer_inline_query(bot, query).await {
|
||||
Ok(true) => {}
|
||||
// No results produced (or nothing to answer): let a repeat of the
|
||||
// same query retry the fetch.
|
||||
Ok(false) | Err(_) => INLINE_DEBOUNCE_STATE.lock().release(user_id, &query_text),
|
||||
}
|
||||
});
|
||||
respond(())
|
||||
}
|
||||
|
||||
/// Fetches the post behind an inline query and answers it. The caller has
|
||||
/// already applied the debounce. Returns `true` when an answer was sent.
|
||||
async fn answer_inline_query(bot: Bot, query: InlineQuery) -> Result<bool, RequestError> {
|
||||
log::debug!(
|
||||
"inline query: {} [key={}]",
|
||||
query.query,
|
||||
log_key(&query.query)
|
||||
);
|
||||
// No retries: the debounce plus a 1s/2s backoff would outlast the inline
|
||||
// query the answer belongs to.
|
||||
match x_media::site::fetch_once(&query.query).await {
|
||||
Ok(Some(fetched)) => {
|
||||
let mut results: Vec<InlineQueryResult> = Vec::new();
|
||||
// Inline results have the same 1024-char caption limit as regular
|
||||
// messages; truncate once here for all items, then apply the same
|
||||
// long-post quoting as the send paths. `answer_inline_query` has no
|
||||
// `AppContext` (the debounce spawns it), so the parsed config comes
|
||||
// from the process-wide static, and the text is the *escaped*
|
||||
// title/content the built-in caption embeds (the raw
|
||||
// `Fetched.title`/`content` differ whenever the post contains
|
||||
// `<`/`&`).
|
||||
let caption = x_media::site::truncate_caption(&fetched.caption);
|
||||
let text = fetched
|
||||
.render_fields()
|
||||
.map(|(_, _, title, content, _)| x_media::site::compose_text(title, content))
|
||||
.unwrap_or_default();
|
||||
let caption = crate::send::quote_long_caption(
|
||||
&caption,
|
||||
&text,
|
||||
super::CONFIG.caption_quote_text_chars,
|
||||
);
|
||||
for (i, media) in fetched.media.iter().enumerate() {
|
||||
let id = format!("{i}");
|
||||
let Some(url) = url::Url::parse(media.url()).ok() else {
|
||||
continue;
|
||||
};
|
||||
let thumbnail = media
|
||||
.thumbnail_url()
|
||||
.and_then(|t| url::Url::parse(t).ok())
|
||||
.unwrap_or_else(|| url.clone());
|
||||
let caption = caption.clone().into_owned();
|
||||
let result = match media {
|
||||
Media::Illustration { .. } => {
|
||||
// Inline photo results have their own (smaller) size
|
||||
// cap; use the reduced variant when one exists.
|
||||
let photo_url = media
|
||||
.smaller_url()
|
||||
.and_then(|u| url::Url::parse(u).ok())
|
||||
.unwrap_or_else(|| url.clone());
|
||||
InlineQueryResult::Photo(
|
||||
InlineQueryResultPhoto::new(id, photo_url, thumbnail)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html),
|
||||
)
|
||||
}
|
||||
Media::Video { .. } => InlineQueryResult::Video(
|
||||
InlineQueryResultVideo::new(
|
||||
id,
|
||||
url,
|
||||
"video/mp4".parse().expect("valid mime"),
|
||||
thumbnail,
|
||||
fetched.title.clone(),
|
||||
)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html),
|
||||
),
|
||||
Media::Animated { .. } => InlineQueryResult::Mpeg4Gif(
|
||||
InlineQueryResultMpeg4Gif::new(id, url, thumbnail)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html),
|
||||
),
|
||||
};
|
||||
results.push(result);
|
||||
}
|
||||
if !results.is_empty() {
|
||||
// Explicit cache window: repeats of the same query within 5
|
||||
// minutes are served by Telegram without hitting the bot.
|
||||
bot.answer_inline_query(query.id, results)
|
||||
.cache_time(300)
|
||||
.await?;
|
||||
return Ok(true);
|
||||
}
|
||||
}
|
||||
Ok(None) => {}
|
||||
Err(e) => log::error!("inline fetch {}: {e}", query.query),
|
||||
}
|
||||
Ok(false)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::DebounceStates;
|
||||
|
||||
const URL_A: &str = "https://x.com/a/status/1";
|
||||
const URL_B: &str = "https://x.com/b/status/2";
|
||||
|
||||
#[test]
|
||||
fn debounce_state_is_per_user() {
|
||||
let mut states = DebounceStates::default();
|
||||
// Two users query different links: both proceed, and neither timer
|
||||
// cancels the other (a single shared slot dropped one of them).
|
||||
assert!(states.note(1, URL_A));
|
||||
assert!(states.note(2, URL_B));
|
||||
assert!(states.claim(1, URL_A), "user 1's answer was cancelled");
|
||||
assert!(states.claim(2, URL_B), "user 2's answer was cancelled");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn answered_query_is_suppressed_per_user_only() {
|
||||
let mut states = DebounceStates::default();
|
||||
assert!(states.note(1, URL_A));
|
||||
assert!(states.claim(1, URL_A));
|
||||
// A repeat of the answered query by the same user is left to
|
||||
// Telegram's inline cache.
|
||||
assert!(!states.note(1, URL_A));
|
||||
// Another user pasting the same link still gets an answer.
|
||||
assert!(states.note(2, URL_A));
|
||||
assert!(states.claim(2, URL_A));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn newer_query_supersedes_and_failed_answer_is_released() {
|
||||
let mut states = DebounceStates::default();
|
||||
assert!(states.note(1, URL_A));
|
||||
assert!(states.note(1, URL_B));
|
||||
// The stale timer for the half-typed query gives up…
|
||||
assert!(!states.claim(1, URL_A));
|
||||
// …and the newest one answers.
|
||||
assert!(states.claim(1, URL_B));
|
||||
// No results → release so a repeat may retry the fetch.
|
||||
states.release(1, URL_B);
|
||||
assert!(states.claim(1, URL_B));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,274 @@
|
||||
//! Update handlers and the per-URL media pipeline.
|
||||
//!
|
||||
//! Split into per-concern modules: [`commands`] (the `/`-command executor),
|
||||
//! [`urls`] (URL extraction + the bounded worker pool + send dispatch),
|
||||
//! [`inline`] (debounced inline queries), [`callback`] (edit-before-forward
|
||||
//! buttons) and [`statics`] (the shared process-wide stores). This module
|
||||
//! holds the message entry point and the helpers the others share.
|
||||
|
||||
mod callback;
|
||||
mod commands;
|
||||
mod inline;
|
||||
mod statics;
|
||||
mod urls;
|
||||
|
||||
pub use callback::callback_query_handler;
|
||||
pub use commands::register_commands;
|
||||
pub use inline::inline_query_handler;
|
||||
pub use statics::{CHAT_STORE, CONFIG, LINK_CACHE, TASK_QUEUE};
|
||||
pub use urls::{start_url_workers, stop_url_workers};
|
||||
|
||||
use crate::ctx::AppContext;
|
||||
use crate::media_sender::MediaSender;
|
||||
use commands::{Command, execute_command};
|
||||
use teloxide::RequestError;
|
||||
use teloxide::prelude::*;
|
||||
use teloxide::types::{ChatId, ChatKind, Message, MessageId, ParseMode, ReplyParameters};
|
||||
use teloxide::utils::command::BotCommands;
|
||||
use urls::{URL_JOBS, extract_urls};
|
||||
|
||||
/// Reply to a message by id, keeping the reply decoration even if the
|
||||
/// original was already deleted. Returns the reply's message id.
|
||||
pub(crate) async fn reply(
|
||||
sender: &dyn MediaSender,
|
||||
chat_id: i64,
|
||||
reply_to: MessageId,
|
||||
text: impl Into<String>,
|
||||
) -> Result<i64, RequestError> {
|
||||
sender
|
||||
.send_message(ChatId(chat_id), text.into(), Some(reply_to), None)
|
||||
.await
|
||||
}
|
||||
|
||||
/// Reply to a message by id with HTML parse mode (same reply decoration as
|
||||
/// [`reply`]). Used by `/test`, whose report is an HTML message (the caption
|
||||
/// is wrapped in a `<blockquote>` to show it exactly as it will render).
|
||||
pub(crate) async fn reply_html(
|
||||
bot: &Bot,
|
||||
chat_id: i64,
|
||||
reply_to: MessageId,
|
||||
text: String,
|
||||
) -> Result<i64, RequestError> {
|
||||
// `<Bot as Requester>::` disambiguates from the MediaSender trait's
|
||||
// same-named method (see media_sender.rs).
|
||||
<Bot as Requester>::send_message(bot, ChatId(chat_id), text)
|
||||
.parse_mode(ParseMode::Html)
|
||||
.reply_parameters(ReplyParameters::new(reply_to).allow_sending_without_reply())
|
||||
.await
|
||||
.map(|message| message.id.0 as i64)
|
||||
}
|
||||
|
||||
/// Log prefix tying the whole lifecycle of one link (fetch → send → cache →
|
||||
/// forward) together: the normalized cache key (`twitter:123…`, `pixiv:123`,
|
||||
/// `bsky:handle/rkey`, `bilibili:123…`) instead of the raw URL, so logs stay
|
||||
/// short and do not echo full user-submitted URLs at info level.
|
||||
pub fn log_key(url: &str) -> String {
|
||||
x_media::site::cache_key(url).unwrap_or_else(|| "<unsupported>".to_string())
|
||||
}
|
||||
|
||||
/// Edit-before-forward: a reply to the prompt swaps the caption of the first
|
||||
/// forwarded message. Returns true when the message was consumed as an edit.
|
||||
/// Body of [`message_handler`]'s edit branch, without teloxide update types so
|
||||
/// it can be driven by tests.
|
||||
async fn edit_message_handler(
|
||||
ctx: &AppContext<'_>,
|
||||
chat_id: i64,
|
||||
reply_to_message_id: i64,
|
||||
text: &str,
|
||||
) -> bool {
|
||||
let chat_data = ctx.chat_store.get(chat_id).await;
|
||||
let Some(edit) = chat_data.edit_message.get(&reply_to_message_id) else {
|
||||
return false;
|
||||
};
|
||||
let Some(first_forward_id) = edit.forward_message_ids.first() else {
|
||||
return false;
|
||||
};
|
||||
let link = format!(
|
||||
"<a href=\"{0}\">{1}</a>",
|
||||
html_escape::encode_double_quoted_attribute(&edit.url),
|
||||
html_escape::encode_text(text)
|
||||
);
|
||||
let new_text = if edit.template.is_empty() {
|
||||
link
|
||||
} else {
|
||||
chat_data
|
||||
.template
|
||||
.get(&edit.template)
|
||||
.map(|template| template.replace("[]", &link))
|
||||
.unwrap_or(link)
|
||||
};
|
||||
match ctx
|
||||
.sender
|
||||
.edit_message_caption(
|
||||
ChatId(chat_id),
|
||||
MessageId(*first_forward_id as i32),
|
||||
new_text,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(()) => log::info!(
|
||||
"edit-before-forward: caption swapped on message {first_forward_id} for prompt {reply_to_message_id}"
|
||||
),
|
||||
Err(e) => log::error!("edit_message_caption failed: {e}"),
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
pub async fn message_handler(bot: Bot, message: Message) -> Result<(), RequestError> {
|
||||
let is_private = matches!(message.chat.kind, ChatKind::Private(_));
|
||||
let sender = message
|
||||
.from
|
||||
.as_ref()
|
||||
.map(|from| from.full_name())
|
||||
.unwrap_or_else(|| "unknown".to_string());
|
||||
let text_preview = message
|
||||
.text()
|
||||
.map(|t| {
|
||||
let end = t.floor_char_boundary(120.min(t.len()));
|
||||
&t[..end]
|
||||
})
|
||||
.unwrap_or("<no text>");
|
||||
// Per-request detail: debug only (message text is user data).
|
||||
log::debug!(
|
||||
"message from {sender} in {} (private={is_private}): {text_preview}",
|
||||
message.chat.id
|
||||
);
|
||||
// URL/edit flows only run in private chats; commands run in any chat.
|
||||
if is_private
|
||||
&& let Some(reply) = message.reply_to_message()
|
||||
&& let Some(text) = message.text()
|
||||
&& edit_message_handler(
|
||||
&AppContext::from_statics(&bot),
|
||||
message.chat.id.0,
|
||||
reply.id.0 as i64,
|
||||
text,
|
||||
)
|
||||
.await
|
||||
{
|
||||
return respond(());
|
||||
}
|
||||
if let Some(text) = message.text()
|
||||
&& let Ok(command) = Command::parse(text, "")
|
||||
{
|
||||
log::debug!("command from {}: {text_preview}", message.chat.id);
|
||||
execute_command(&bot, &message, command).await?;
|
||||
return respond(());
|
||||
}
|
||||
if is_private {
|
||||
let urls = extract_urls(&message);
|
||||
if !urls.is_empty() {
|
||||
// Debug only, and echo the normalized keys instead of the raw URLs.
|
||||
let keys: Vec<String> = urls.iter().map(|u| log_key(u)).collect();
|
||||
log::debug!("extracted {} URL(s): {keys:?}", urls.len());
|
||||
}
|
||||
for url in urls {
|
||||
// Clone out of the lock: the parking_lot guard is !Send and must
|
||||
// not be held across the await below.
|
||||
let Some(tx) = URL_JOBS.lock().clone() else {
|
||||
log::warn!("url workers not started; dropping link");
|
||||
break;
|
||||
};
|
||||
// A closed channel means the workers are stopping (shutdown):
|
||||
// report the dropped link instead of losing it silently.
|
||||
if tx.send((message.clone(), url)).await.is_err() {
|
||||
log::warn!("url workers stopped; dropping link");
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
respond(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::ctx::test_support::TestStores;
|
||||
use crate::media_sender::test_support::{MockSender, Outcome};
|
||||
use crate::state::EditMessage;
|
||||
use teloxide::ApiError;
|
||||
|
||||
const PROMPT_ID: i64 = 7;
|
||||
const FORWARDED_ID: i64 = 9;
|
||||
|
||||
fn api_error() -> RequestError {
|
||||
RequestError::Api(ApiError::Unknown("Bad Request: message not found".into()))
|
||||
}
|
||||
|
||||
/// Seeds a prompt record; `template` names the chat template used for it
|
||||
/// (empty = none, the caption gets the bare link).
|
||||
async fn seed_prompt(ctx: &AppContext<'_>, template: &str) {
|
||||
ctx.chat_store
|
||||
.update(1, |data| {
|
||||
data.template
|
||||
.insert("tpl".to_string(), "<b>[]</b>".to_string());
|
||||
data.edit_message.insert(
|
||||
PROMPT_ID,
|
||||
EditMessage {
|
||||
url: "https://x.com/u/status/1".into(),
|
||||
chat_id: 1,
|
||||
forward_message_ids: vec![FORWARDED_ID],
|
||||
template: template.to_string(),
|
||||
created_at: crate::db::unix_now(),
|
||||
},
|
||||
);
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn reply_to_a_prompt_swaps_the_caption_through_its_template() {
|
||||
let sender = MockSender::scripted(vec![Outcome::EditOk], api_error);
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
seed_prompt(&ctx, "tpl").await;
|
||||
|
||||
let consumed = edit_message_handler(&ctx, 1, PROMPT_ID, "new caption").await;
|
||||
|
||||
assert!(consumed, "a reply to the prompt must be consumed");
|
||||
assert_eq!(
|
||||
sender.captions(),
|
||||
vec!["<b><a href=\"https://x.com/u/status/1\">new caption</a></b>"]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn reply_text_and_url_are_escaped_into_the_caption() {
|
||||
let sender = MockSender::scripted(vec![Outcome::EditOk], api_error);
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
seed_prompt(&ctx, "").await;
|
||||
|
||||
edit_message_handler(&ctx, 1, PROMPT_ID, "<script>alert(1)</script>").await;
|
||||
|
||||
// No raw markup from user text may reach the HTML caption.
|
||||
assert_eq!(
|
||||
sender.captions(),
|
||||
vec!["<a href=\"https://x.com/u/status/1\"><script>alert(1)</script></a>"]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn a_failed_caption_swap_still_consumes_the_reply() {
|
||||
let sender = MockSender::scripted(vec![Outcome::EditErr], api_error);
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
seed_prompt(&ctx, "tpl").await;
|
||||
|
||||
// The edit failed (message deleted etc.); the reply must still be
|
||||
// swallowed instead of being treated as a link to fetch.
|
||||
assert!(edit_message_handler(&ctx, 1, PROMPT_ID, "new caption").await);
|
||||
assert_eq!(sender.calls(), vec!["edit_message_caption"]);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn reply_to_an_unrelated_message_is_not_consumed() {
|
||||
let sender = MockSender::scripted(vec![], api_error);
|
||||
let stores = TestStores::new();
|
||||
let ctx = stores.ctx(&sender);
|
||||
|
||||
// No prompt record for that message id → the reply runs the normal
|
||||
// (URL/command) path instead.
|
||||
assert!(!edit_message_handler(&ctx, 1, PROMPT_ID, "hello").await);
|
||||
assert!(sender.calls().is_empty());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
//! Process-wide singletons shared by the handler modules: the one SQLite
|
||||
//! pool (and the three stores built on it) plus the configuration.
|
||||
|
||||
use crate::config::Config;
|
||||
use crate::db::{self};
|
||||
use crate::link_cache::LinkCache;
|
||||
use crate::queue::PersistentTaskQueue;
|
||||
use crate::state::ChatStore;
|
||||
use std::sync::{Arc, LazyLock};
|
||||
|
||||
/// One shared SQLite pool for the three stores (chat state, task queue, link
|
||||
/// cache): a single pool bounds concurrent DB work on `data/task_queue.db`
|
||||
/// instead of three independent pools competing for the same file. The schema
|
||||
/// for all three tables is initialized once, here.
|
||||
static DB: LazyLock<Arc<db::DbPool>> = LazyLock::new(|| {
|
||||
let path = db_path();
|
||||
db::open_store(&path.to_string_lossy()).expect("failed to open database")
|
||||
});
|
||||
|
||||
/// DB file location: `$DATA_DIR/task_queue.db` (default `data`, relative to
|
||||
/// the working directory — keeps the docker-compose `./data` mount and local
|
||||
/// runs unchanged). The directory is created if missing: SQLite does not
|
||||
/// create parent dirs, so the old hardcoded `data/task_queue.db` failed with
|
||||
/// a confusing error when started from a directory without `data/`, and a
|
||||
/// CWD-relative path is a footgun for systemd / cron deployments — `DATA_DIR`
|
||||
/// lets them pin the state anywhere.
|
||||
fn db_path() -> std::path::PathBuf {
|
||||
let dir = std::env::var("DATA_DIR").unwrap_or_else(|_| "data".to_string());
|
||||
let dir_path = std::path::Path::new(&dir);
|
||||
std::fs::create_dir_all(dir_path).expect("failed to create data directory");
|
||||
dir_path.join("task_queue.db")
|
||||
}
|
||||
|
||||
pub static CHAT_STORE: LazyLock<ChatStore> = LazyLock::new(|| ChatStore::new(Arc::clone(&DB)));
|
||||
pub static TASK_QUEUE: LazyLock<PersistentTaskQueue> =
|
||||
LazyLock::new(|| PersistentTaskQueue::new(Arc::clone(&DB)));
|
||||
pub static LINK_CACHE: LazyLock<LinkCache> = LazyLock::new(|| LinkCache::new(Arc::clone(&DB)));
|
||||
pub static CONFIG: LazyLock<Config> = LazyLock::new(Config::load);
|
||||
@@ -0,0 +1,698 @@
|
||||
//! URL extraction and the per-URL media pipeline: bounded job channel +
|
||||
//! worker pool, link-cache fast path, fetch, task build and send dispatch.
|
||||
|
||||
use super::{log_key, reply};
|
||||
use crate::ctx::{AppContext, CONTEXT};
|
||||
use crate::link_cache::{CachedMediaKind, CachedPost};
|
||||
use crate::send::{self, MediaItemPayload, Task};
|
||||
use crate::state::ChatData;
|
||||
use std::collections::HashSet;
|
||||
use std::sync::LazyLock;
|
||||
use teloxide::types::{ChatAction, ChatId, Message, MessageEntityKind, MessageId};
|
||||
use x_media::media::Media;
|
||||
|
||||
/// One URL job: the message + the extracted URL (the sender and stores come
|
||||
/// from the shared [`AppContext`], assembled from statics inside the worker).
|
||||
type UrlJob = (Message, String);
|
||||
/// Bounded channel of URL jobs drained by [`start_url_workers`]. The bound
|
||||
/// caps both queued memory and shutdown backlog; a full channel applies
|
||||
/// backpressure to the per-chat handler instead of spawning unbounded tasks.
|
||||
pub(crate) static URL_JOBS: LazyLock<
|
||||
parking_lot::Mutex<Option<tokio::sync::mpsc::Sender<UrlJob>>>,
|
||||
> = LazyLock::new(|| parking_lot::Mutex::new(None));
|
||||
/// Set by main's shutdown sequence; workers stop pulling new jobs.
|
||||
pub(crate) static URL_STOP: std::sync::atomic::AtomicBool =
|
||||
std::sync::atomic::AtomicBool::new(false);
|
||||
|
||||
/// JoinHandles of the URL workers, awaited by [`stop_url_workers`].
|
||||
static URL_WORKER_HANDLES: LazyLock<parking_lot::Mutex<Option<Vec<tokio::task::JoinHandle<()>>>>> =
|
||||
LazyLock::new(|| parking_lot::Mutex::new(None));
|
||||
|
||||
/// Worker count draining URL jobs; keeps the old 8-permit concurrency cap
|
||||
/// while bounding how many jobs can be queued at all.
|
||||
const URL_WORKERS: usize = 8;
|
||||
|
||||
/// Starts the URL job workers (called once from main after the queue starts).
|
||||
/// teloxide dispatches updates to a per-chat worker that handles them
|
||||
/// sequentially, so a batch-forward of many messages would otherwise be
|
||||
/// processed one at a time (fetch + send each, roughly a second per
|
||||
/// message); the workers add throughput, and FIFO order preserves per-message
|
||||
/// URL order.
|
||||
pub async fn start_url_workers() {
|
||||
let (tx, rx) = tokio::sync::mpsc::channel::<UrlJob>(256);
|
||||
*URL_JOBS.lock() = Some(tx);
|
||||
let rx = std::sync::Arc::new(tokio::sync::Mutex::new(rx));
|
||||
let mut handles = Vec::with_capacity(URL_WORKERS);
|
||||
for _ in 0..URL_WORKERS {
|
||||
let rx = std::sync::Arc::clone(&rx);
|
||||
handles.push(tokio::spawn(async move {
|
||||
while !URL_STOP.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
let job = rx.lock().await.recv().await;
|
||||
match job {
|
||||
Some((message, url)) => {
|
||||
url_media(
|
||||
&CONTEXT,
|
||||
message.chat.id.0,
|
||||
message.id.0 as i64,
|
||||
&url,
|
||||
PostSend::FromChat,
|
||||
)
|
||||
.await
|
||||
}
|
||||
None => break,
|
||||
}
|
||||
}
|
||||
}));
|
||||
}
|
||||
*URL_WORKER_HANDLES.lock() = Some(handles);
|
||||
}
|
||||
|
||||
/// Stops the URL workers: sets the stop flag, drops the job channel (so
|
||||
/// workers blocked in \`recv()\` wake with \`None\` and exit) and awaits the
|
||||
/// worker tasks. Each worker finishes its in-flight job first; jobs still
|
||||
/// queued in the channel are abandoned (the old implementation neither
|
||||
/// drained them nor woke blocked workers — it only set a flag checked
|
||||
/// between jobs).
|
||||
pub async fn stop_url_workers() {
|
||||
URL_STOP.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
// Dropping the sender makes every worker's recv() return None.
|
||||
*URL_JOBS.lock() = None;
|
||||
// Take the handles first so the lock guard drops before the awaits.
|
||||
let handles = URL_WORKER_HANDLES.lock().take();
|
||||
if let Some(handles) = handles {
|
||||
for handle in handles {
|
||||
let _ = handle.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Extracts URL and text-link entities (text + caption), deduped in order.
|
||||
pub fn extract_urls(message: &Message) -> Vec<String> {
|
||||
let mut urls = Vec::new();
|
||||
for entity in message.parse_entities().into_iter().flatten() {
|
||||
match entity.kind() {
|
||||
MessageEntityKind::Url => urls.push(entity.text().to_string()),
|
||||
MessageEntityKind::TextLink { url } => urls.push(url.to_string()),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
for entity in message.parse_caption_entities().into_iter().flatten() {
|
||||
match entity.kind() {
|
||||
MessageEntityKind::Url => urls.push(entity.text().to_string()),
|
||||
MessageEntityKind::TextLink { url } => urls.push(url.to_string()),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
let mut seen = HashSet::new();
|
||||
// Dedup by the normalized post id so variant URLs of the same post
|
||||
// (/status/1 vs /status/1/photo/1) are sent once; unsupported URLs fall
|
||||
// back to exact-string dedup.
|
||||
urls.retain(|url| seen.insert(x_media::site::cache_key(url).unwrap_or_else(|| url.clone())));
|
||||
urls
|
||||
}
|
||||
|
||||
/// For locally produced media (encoded ugoira MP4) the thumbnail URL is a
|
||||
/// hotlink-protected remote URL Telegram may not fetch; let Telegram generate
|
||||
/// its own thumbnail instead.
|
||||
fn thumbnail_for(media: &Media) -> Option<String> {
|
||||
let url = media.url();
|
||||
if url.starts_with("http://") || url.starts_with("https://") {
|
||||
// An empty thumbnail string (misskey video/gif files without a
|
||||
// thumbnailUrl) must not reach Telegram; let it generate its own.
|
||||
media
|
||||
.thumbnail_url()
|
||||
.map(str::to_string)
|
||||
.filter(|t| !t.is_empty())
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn media_to_payload(media: &Media, sensitive: bool) -> MediaItemPayload {
|
||||
let fallback_url = media.smaller_url().map(str::to_string);
|
||||
match media {
|
||||
// A gif inside a group becomes a video item; a lone gif takes the
|
||||
// animation path (see url_media).
|
||||
Media::Illustration { .. } => MediaItemPayload::Photo {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
Media::Video { .. } => MediaItemPayload::Video {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
thumbnail: thumbnail_for(media),
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
Media::Animated { .. } => MediaItemPayload::Video {
|
||||
media: media.url().to_string(),
|
||||
has_spoiler: sensitive,
|
||||
thumbnail: thumbnail_for(media),
|
||||
fallback_url,
|
||||
file_id: false,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// Sends a task and handles the outcome: post-send actions on success, retry
|
||||
/// enqueue on retryable failure, reply + link-cache invalidation on
|
||||
/// permanent failure (a stale cached file id must not repeat forever).
|
||||
async fn dispatch_send(
|
||||
ctx: &AppContext<'_>,
|
||||
chat_id: i64,
|
||||
reply_to: MessageId,
|
||||
task: &Task,
|
||||
url: &str,
|
||||
) {
|
||||
let result = match task {
|
||||
Task::SendAnimation { .. } => send::send_animation(ctx, task).await,
|
||||
Task::SendMediaSequence { .. } => send::send_media_sequence(ctx, task).await,
|
||||
Task::ForwardMessages { .. } => unreachable!(),
|
||||
};
|
||||
match result {
|
||||
Ok(message_ids) => {
|
||||
log::info!(
|
||||
"sent {} message(s) for [key={}]",
|
||||
message_ids.len(),
|
||||
log_key(url)
|
||||
);
|
||||
send::post_send_actions(ctx, task, message_ids).await;
|
||||
send::settle_task(ctx, task, send::Settled::Sent).await;
|
||||
}
|
||||
Err(send::SendError::Retryable {
|
||||
delay_seconds,
|
||||
task,
|
||||
}) => {
|
||||
log::info!(
|
||||
"send for [key={}] failed, queued for retry in {delay_seconds:.1}s",
|
||||
log_key(url)
|
||||
);
|
||||
send::enqueue_retry(ctx.task_queue, *task, delay_seconds).await;
|
||||
let _ = reply(
|
||||
ctx.sender,
|
||||
chat_id,
|
||||
reply_to,
|
||||
"Send failed. Task queued for retry.",
|
||||
)
|
||||
.await;
|
||||
}
|
||||
Err(send::SendError::Permanent {
|
||||
message: err_message,
|
||||
task,
|
||||
}) => {
|
||||
send::settle_task(ctx, &task, send::Settled::Failed).await;
|
||||
log::error!("send for {url} failed permanently: {err_message}");
|
||||
let _ = reply(
|
||||
ctx.sender,
|
||||
chat_id,
|
||||
reply_to,
|
||||
format!("Send failed: {err_message}"),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a send also runs the chat's post-send actions. `/test` sends with
|
||||
/// them suppressed so a test can never forward to the channel or open the
|
||||
/// edit-before-forward prompt; a normal link uses whatever the chat is
|
||||
/// configured with.
|
||||
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
|
||||
pub(crate) enum PostSend {
|
||||
/// Apply the chat's `forward_channel_id` / `edit_before_forward`.
|
||||
FromChat,
|
||||
/// Send only: no channel forward, no edit prompt.
|
||||
Suppressed,
|
||||
}
|
||||
|
||||
/// Builds the send task from ready-made items, sharing the payload shape
|
||||
/// between the fresh-fetch, link-cache and `/test` paths.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn build_send_task(
|
||||
chat_data: &ChatData,
|
||||
chat_id: i64,
|
||||
reply_to_message_id: i64,
|
||||
source_url: String,
|
||||
caption: String,
|
||||
items: Vec<MediaItemPayload>,
|
||||
cache_data: Option<CachedPost>,
|
||||
post_send: PostSend,
|
||||
) -> Task {
|
||||
// Notification ids stay set in both modes: a queued retry that
|
||||
// dead-letters should still tell the chat.
|
||||
let (edit_before_forward, forward_channel_id) = match post_send {
|
||||
PostSend::FromChat => (chat_data.edit_before_forward, chat_data.forward_channel_id),
|
||||
PostSend::Suppressed => (false, None),
|
||||
};
|
||||
if items.len() == 1 && matches!(items[0], MediaItemPayload::Animation { .. }) {
|
||||
Task::SendAnimation {
|
||||
chat_id,
|
||||
reply_to_message_id,
|
||||
caption,
|
||||
animation: items.into_iter().next().unwrap(),
|
||||
source_url,
|
||||
edit_before_forward,
|
||||
forward_channel_id,
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(reply_to_message_id),
|
||||
cache_data,
|
||||
}
|
||||
} else {
|
||||
Task::SendMediaSequence {
|
||||
chat_id,
|
||||
reply_to_message_id,
|
||||
caption,
|
||||
// Photos first so a mixed photo+video group starts with a photo
|
||||
// (Telegram's sendMediaGroup rule); order within each kind is kept.
|
||||
media_batches: send::chunk_media_items(send::photos_first(items)),
|
||||
batch_index: 0,
|
||||
sent_message_ids: vec![],
|
||||
source_url,
|
||||
edit_before_forward,
|
||||
forward_channel_id,
|
||||
notify_chat_id: Some(chat_id),
|
||||
notify_message_id: Some(reply_to_message_id),
|
||||
cache_data,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The per-URL pipeline: link cache → fetch → build → send → post-send.
|
||||
///
|
||||
/// `post_send` selects whether the chat's forward/edit settings apply: the URL
|
||||
/// workers pass [`PostSend::FromChat`], the `/test` command
|
||||
/// [`PostSend::Suppressed`]. Everything else (cache write, retry enqueue,
|
||||
/// dead-letter notification) is identical.
|
||||
pub(crate) async fn url_media(
|
||||
ctx: &AppContext<'_>,
|
||||
chat_id: i64,
|
||||
reply_to_message_id: i64,
|
||||
url: &str,
|
||||
post_send: PostSend,
|
||||
) {
|
||||
let reply_to = MessageId(reply_to_message_id as i32);
|
||||
if let Err(e) = ctx
|
||||
.sender
|
||||
.send_chat_action(ChatId(chat_id), ChatAction::Typing)
|
||||
.await
|
||||
{
|
||||
log::error!("send_chat_action failed: {e}");
|
||||
}
|
||||
|
||||
// Link cache: a post sent before is re-sent from Telegram file ids —
|
||||
// no source-site request, no download, no upload. Keyed by the
|
||||
// normalized post id so x.com / fxtwitter / /photo/N variants collide.
|
||||
if let Some(key) = x_media::site::cache_key(url)
|
||||
&& let Some(cached) = ctx.link_cache.get(&key, ctx.config.link_cache_ttl).await
|
||||
{
|
||||
log::debug!("link cache hit for {key}");
|
||||
let chat_data = ctx.chat_store.get(chat_id).await;
|
||||
// Cache keys are prefixed with the site id ("twitter:…"), matching
|
||||
// the value a fresh fetch would read from Fetched::site_id.
|
||||
let site = x_media::site::site_id_from_key(&key);
|
||||
let format = chat_data
|
||||
.message_format
|
||||
.get(site)
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let caption = if format.is_empty() {
|
||||
x_media::site::truncate_caption(&cached.caption)
|
||||
} else {
|
||||
x_media::site::caption_from_fields(
|
||||
&format,
|
||||
"",
|
||||
&cached.url,
|
||||
&cached.author,
|
||||
&cached.author_url,
|
||||
&cached.title,
|
||||
&cached.content,
|
||||
&cached.tags,
|
||||
)
|
||||
};
|
||||
let items: Vec<MediaItemPayload> = cached
|
||||
.media
|
||||
.iter()
|
||||
.map(|m| match m.kind {
|
||||
CachedMediaKind::Photo => MediaItemPayload::Photo {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
fallback_url: None,
|
||||
file_id: true,
|
||||
},
|
||||
CachedMediaKind::Video => MediaItemPayload::Video {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
thumbnail: None,
|
||||
fallback_url: None,
|
||||
file_id: true,
|
||||
},
|
||||
CachedMediaKind::Animation => MediaItemPayload::Animation {
|
||||
media: m.file_id.clone(),
|
||||
has_spoiler: cached.sensitive,
|
||||
file_id: true,
|
||||
},
|
||||
})
|
||||
.collect();
|
||||
let task = build_send_task(
|
||||
&chat_data,
|
||||
chat_id,
|
||||
reply_to_message_id,
|
||||
cached.url.clone(),
|
||||
caption,
|
||||
items,
|
||||
Some(cached),
|
||||
post_send,
|
||||
);
|
||||
dispatch_send(ctx, chat_id, reply_to, &task, url).await;
|
||||
return;
|
||||
}
|
||||
|
||||
log::debug!("fetching {url} [key={}]", log_key(url));
|
||||
match x_media::site::fetch(url).await {
|
||||
// Unsupported links are ignored silently (Python parity).
|
||||
Ok(None) => {
|
||||
log::debug!("no site pattern matches {url}; ignoring");
|
||||
}
|
||||
// Retries exhausted: notify the user (Rust-only requirement 3).
|
||||
Err(e) => {
|
||||
log::error!("fetch {url}: {e}");
|
||||
let _ = reply(
|
||||
ctx.sender,
|
||||
chat_id,
|
||||
reply_to,
|
||||
"Failed to fetch media from this link.",
|
||||
)
|
||||
.await;
|
||||
}
|
||||
Ok(Some(mut fetched)) => {
|
||||
if fetched.media.is_empty() {
|
||||
let _ = reply(
|
||||
ctx.sender,
|
||||
chat_id,
|
||||
reply_to,
|
||||
"No media found or media type is not supported.",
|
||||
)
|
||||
.await;
|
||||
return;
|
||||
}
|
||||
let chat_data = ctx.chat_store.get(chat_id).await;
|
||||
// Per-site caption format override (empty -> built-in caption).
|
||||
let format = chat_data
|
||||
.message_format
|
||||
.get(fetched.site_name())
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let caption = fetched.caption_with(&format);
|
||||
// Raw render data for the link cache; the send fills in the
|
||||
// Telegram file ids and persists the entry.
|
||||
let cache_data =
|
||||
fetched
|
||||
.render_fields()
|
||||
.map(|(author, author_url, title, content, tags)| CachedPost {
|
||||
url: fetched.source_url.clone(),
|
||||
caption: fetched.caption.clone(),
|
||||
title: title.to_string(),
|
||||
content: content.to_string(),
|
||||
author: author.to_string(),
|
||||
author_url: author_url.to_string(),
|
||||
tags: tags.to_string(),
|
||||
sensitive: fetched.sensitive,
|
||||
media: vec![],
|
||||
});
|
||||
let items: Vec<MediaItemPayload> = fetched
|
||||
.media
|
||||
.iter()
|
||||
.map(|media| media_to_payload(media, fetched.sensitive))
|
||||
.collect();
|
||||
let task = build_send_task(
|
||||
&chat_data,
|
||||
chat_id,
|
||||
reply_to_message_id,
|
||||
fetched.source_url.clone(),
|
||||
caption,
|
||||
items,
|
||||
cache_data,
|
||||
post_send,
|
||||
);
|
||||
// Hand the keep-alive temp dir (ugoira / bsky remux MP4) to the
|
||||
// retry registry: a queued retry runs after this function returns
|
||||
// and the fetch's own TempDir is dropped, so without this the
|
||||
// local file would be gone by the time the retry sends it.
|
||||
if let Some(dir) = fetched.take_keep_alive() {
|
||||
send::KEEP_ALIVE.lock().push(dir);
|
||||
}
|
||||
dispatch_send(ctx, chat_id, reply_to, &task, url).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::ctx::test_support::TestStores;
|
||||
use crate::link_cache::CachedMedia;
|
||||
use crate::media_sender::test_support::{MockSender, Outcome};
|
||||
use std::time::Duration;
|
||||
use teloxide::{ApiError, RequestError};
|
||||
|
||||
fn permanent_error() -> RequestError {
|
||||
RequestError::Api(ApiError::Unknown(
|
||||
"Bad Request: message is not modified".into(),
|
||||
))
|
||||
}
|
||||
|
||||
fn cached_photo_entry() -> CachedPost {
|
||||
CachedPost {
|
||||
url: "https://x.com/u/status/1".into(),
|
||||
caption: "cap".into(),
|
||||
title: "t".into(),
|
||||
content: "c".into(),
|
||||
author: "a".into(),
|
||||
author_url: "au".into(),
|
||||
tags: "".into(),
|
||||
sensitive: false,
|
||||
media: vec![CachedMedia {
|
||||
kind: CachedMediaKind::Photo,
|
||||
file_id: "file-1".into(),
|
||||
}],
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn cache_hit_sends_file_ids_and_invalidates_on_permanent_failure() {
|
||||
let stores = TestStores::new();
|
||||
let sender = MockSender::scripted(
|
||||
vec![Outcome::GroupErr, Outcome::MessageErr],
|
||||
permanent_error,
|
||||
);
|
||||
let ctx = stores.ctx(&sender);
|
||||
stores
|
||||
.link_cache()
|
||||
.put("twitter:1", &cached_photo_entry())
|
||||
.await;
|
||||
|
||||
url_media(&ctx, 1, 2, "https://x.com/u/status/1", PostSend::FromChat).await;
|
||||
|
||||
// The cached file id went out as a group send; the permanent failure
|
||||
// then triggered the fire-and-forget reply (its mock error is fine).
|
||||
assert_eq!(
|
||||
sender.calls(),
|
||||
vec!["send_chat_action", "send_media_group", "send_message"]
|
||||
);
|
||||
// The stale cache entry was invalidated so the next request re-fetches.
|
||||
assert!(
|
||||
stores
|
||||
.link_cache()
|
||||
.get("twitter:1", Duration::from_secs(3600))
|
||||
.await
|
||||
.is_none()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn cache_hit_success_keeps_the_cache_entry() {
|
||||
let stores = TestStores::new();
|
||||
let sender = MockSender::scripted(vec![Outcome::GroupOk], permanent_error);
|
||||
let ctx = stores.ctx(&sender);
|
||||
stores
|
||||
.link_cache()
|
||||
.put("twitter:1", &cached_photo_entry())
|
||||
.await;
|
||||
|
||||
url_media(&ctx, 1, 2, "https://x.com/u/status/1", PostSend::FromChat).await;
|
||||
|
||||
assert_eq!(sender.calls(), vec!["send_chat_action", "send_media_group"]);
|
||||
// Success must not evict the entry.
|
||||
assert!(
|
||||
stores
|
||||
.link_cache()
|
||||
.get("twitter:1", Duration::from_secs(3600))
|
||||
.await
|
||||
.is_some()
|
||||
);
|
||||
}
|
||||
|
||||
/// The caption-quote threshold matches the post's text inside the caption,
|
||||
/// so a long-text cache hit is quoted and a short-text one is not.
|
||||
#[tokio::test]
|
||||
async fn cache_hit_quotes_a_long_text_caption() {
|
||||
let mut stores = TestStores::new();
|
||||
stores.config_mut().caption_quote_text_chars = 3;
|
||||
let prefix = "https://x.com/u/status/1\n<a href=\"au\">a</a>: ";
|
||||
|
||||
for (text, expected) in [
|
||||
(
|
||||
"abc",
|
||||
format!("{prefix}<blockquote expandable>abc</blockquote>"),
|
||||
),
|
||||
("ab", format!("{prefix}ab")),
|
||||
] {
|
||||
let sender = MockSender::scripted(vec![Outcome::GroupOk], permanent_error);
|
||||
let ctx = stores.ctx(&sender);
|
||||
let mut entry = cached_photo_entry();
|
||||
entry.caption = format!("{prefix}{text}");
|
||||
entry.title = String::new();
|
||||
entry.content = text.into();
|
||||
stores.link_cache().put("twitter:1", &entry).await;
|
||||
|
||||
url_media(&ctx, 1, 2, "https://x.com/u/status/1", PostSend::FromChat).await;
|
||||
|
||||
assert_eq!(sender.captions(), vec![expected], "text {text:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn unsupported_url_is_ignored_silently() {
|
||||
let stores = TestStores::new();
|
||||
let sender = MockSender::scripted(vec![], permanent_error);
|
||||
let ctx = stores.ctx(&sender);
|
||||
|
||||
// No cache key → the fetch dispatcher returns Ok(None) without any
|
||||
// network; nothing is sent or replied.
|
||||
url_media(
|
||||
&ctx,
|
||||
1,
|
||||
2,
|
||||
"https://example.com/not-a-post",
|
||||
PostSend::FromChat,
|
||||
)
|
||||
.await;
|
||||
assert_eq!(sender.calls(), vec!["send_chat_action"]);
|
||||
}
|
||||
|
||||
// ── Send modes: the URL flow vs `/test` ─────────────────────────────
|
||||
|
||||
/// A chat that has both post-send actions configured.
|
||||
async fn seed_post_send_settings(ctx: &AppContext<'_>) {
|
||||
ctx.chat_store
|
||||
.update(1, |data| {
|
||||
data.forward_channel_id = Some(2);
|
||||
data.edit_before_forward = true;
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn chat_settings_apply_to_the_normal_link_flow() {
|
||||
let stores = TestStores::new();
|
||||
let sender =
|
||||
MockSender::scripted(vec![Outcome::GroupOk, Outcome::MessageOk], permanent_error);
|
||||
let ctx = stores.ctx(&sender);
|
||||
stores
|
||||
.link_cache()
|
||||
.put("twitter:1", &cached_photo_entry())
|
||||
.await;
|
||||
seed_post_send_settings(&ctx).await;
|
||||
|
||||
url_media(&ctx, 1, 2, "https://x.com/u/status/1", PostSend::FromChat).await;
|
||||
|
||||
// Media group, then the edit prompt (edit-before-forward wins over the
|
||||
// channel forward, which only runs once the prompt is confirmed).
|
||||
assert_eq!(
|
||||
sender.calls(),
|
||||
vec!["send_chat_action", "send_media_group", "send_message"]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_mode_sends_the_media_without_forwarding_or_editing() {
|
||||
let stores = TestStores::new();
|
||||
// Only the group send is scripted: any forward (copy_messages) or edit
|
||||
// prompt (send_message) would panic with "unexpected outcome".
|
||||
let sender = MockSender::scripted(vec![Outcome::GroupOk], permanent_error);
|
||||
let ctx = stores.ctx(&sender);
|
||||
stores
|
||||
.link_cache()
|
||||
.put("twitter:1", &cached_photo_entry())
|
||||
.await;
|
||||
seed_post_send_settings(&ctx).await;
|
||||
|
||||
url_media(&ctx, 1, 2, "https://x.com/u/status/1", PostSend::Suppressed).await;
|
||||
|
||||
assert_eq!(sender.calls(), vec!["send_chat_action", "send_media_group"]);
|
||||
// The send is otherwise ordinary: the post stays cached.
|
||||
assert!(
|
||||
stores
|
||||
.link_cache()
|
||||
.get("twitter:1", Duration::from_secs(3600))
|
||||
.await
|
||||
.is_some()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn send_mode_decides_whether_chat_actions_ride_along() {
|
||||
let chat = ChatData {
|
||||
forward_channel_id: Some(2),
|
||||
edit_before_forward: true,
|
||||
..ChatData::default()
|
||||
};
|
||||
|
||||
let with_chat = build_send_task(
|
||||
&chat,
|
||||
1,
|
||||
2,
|
||||
"https://x.com/u/status/1".into(),
|
||||
"cap".into(),
|
||||
vec![],
|
||||
None,
|
||||
PostSend::FromChat,
|
||||
);
|
||||
let Task::SendMediaSequence {
|
||||
edit_before_forward,
|
||||
forward_channel_id,
|
||||
..
|
||||
} = with_chat
|
||||
else {
|
||||
panic!("expected a media sequence task");
|
||||
};
|
||||
assert!(edit_before_forward);
|
||||
assert_eq!(forward_channel_id, Some(2));
|
||||
|
||||
let suppressed = build_send_task(
|
||||
&chat,
|
||||
1,
|
||||
2,
|
||||
"https://x.com/u/status/1".into(),
|
||||
"cap".into(),
|
||||
vec![],
|
||||
None,
|
||||
PostSend::Suppressed,
|
||||
);
|
||||
let Task::SendMediaSequence {
|
||||
edit_before_forward,
|
||||
forward_channel_id,
|
||||
notify_chat_id,
|
||||
..
|
||||
} = suppressed
|
||||
else {
|
||||
panic!("expected a media sequence task");
|
||||
};
|
||||
assert!(!edit_before_forward, "`/test` must not open an edit prompt");
|
||||
assert_eq!(forward_channel_id, None, "`/test` must not forward");
|
||||
// Dead-letter notification still reaches the chat that asked.
|
||||
assert_eq!(notify_chat_id, Some(1));
|
||||
}
|
||||
}
|
||||
@@ -5,11 +5,13 @@
|
||||
//! key]. A repeated link is then answered entirely from local state — no
|
||||
//! re-fetch of the source site, no re-upload — and no media file is stored
|
||||
//! on disk (the file ids point at Telegram's servers). Entries expire after
|
||||
//! [`Config::link_cache_ttl`]; a stale entry is dropped lazily on read and
|
||||
//! `Config::link_cache_ttl`; a stale entry is dropped lazily on read and
|
||||
//! by the periodic prune in `main`.
|
||||
|
||||
use rusqlite::{Connection, params};
|
||||
use crate::db::now_f64;
|
||||
use rusqlite::params;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::sync::Arc;
|
||||
use std::time::Duration;
|
||||
|
||||
#[derive(Serialize, Deserialize, Clone, Debug, PartialEq)]
|
||||
@@ -36,6 +38,10 @@ pub struct CachedPost {
|
||||
/// override).
|
||||
pub caption: String,
|
||||
pub title: String,
|
||||
/// The post's body text. Defaulted on read: entries written before the
|
||||
/// title/content split carry it inside `title`.
|
||||
#[serde(default)]
|
||||
pub content: String,
|
||||
pub author: String,
|
||||
pub author_url: String,
|
||||
pub tags: String,
|
||||
@@ -44,24 +50,16 @@ pub struct CachedPost {
|
||||
}
|
||||
|
||||
/// SQLite-backed cache sharing `data/task_queue.db` with the queue and chat
|
||||
/// state (same `open_db` pattern: busy timeout, `spawn_blocking` I/O).
|
||||
/// state (same shared pool, see [`crate::db::open_store`]).
|
||||
pub struct LinkCache {
|
||||
db_path: String,
|
||||
pool: Arc<crate::db::DbPool>,
|
||||
}
|
||||
|
||||
impl LinkCache {
|
||||
pub fn open(db_path: &str) -> Self {
|
||||
if let Ok(conn) = Connection::open(db_path)
|
||||
&& let Err(e) = conn.execute_batch(
|
||||
"CREATE TABLE IF NOT EXISTS link_cache (url TEXT PRIMARY KEY, \
|
||||
payload TEXT NOT NULL, created_at REAL NOT NULL);",
|
||||
)
|
||||
{
|
||||
log::error!("failed to initialize link cache schema: {e}");
|
||||
}
|
||||
Self {
|
||||
db_path: db_path.to_string(),
|
||||
}
|
||||
/// Wraps the shared DB pool (the `link_cache` table lives in the merged
|
||||
/// schema alongside `tasks` and `chat_state`).
|
||||
pub fn new(pool: Arc<crate::db::DbPool>) -> Self {
|
||||
LinkCache { pool }
|
||||
}
|
||||
|
||||
/// Returns the cached post if present and not expired; a stale entry is
|
||||
@@ -69,24 +67,32 @@ impl LinkCache {
|
||||
pub async fn get(&self, key: &str, ttl: Duration) -> Option<CachedPost> {
|
||||
let key = key.to_string();
|
||||
let ttl = ttl.as_secs_f64();
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
let mut stmt =
|
||||
conn.prepare("SELECT payload, created_at FROM link_cache WHERE url = ?1")?;
|
||||
let mut rows = stmt.query(params![key])?;
|
||||
let Some(row) = rows.next()? else {
|
||||
return Ok(None);
|
||||
};
|
||||
let payload: String = row.get(0)?;
|
||||
let created_at: f64 = row.get(1)?;
|
||||
if now_f64() - created_at > ttl {
|
||||
conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key])?;
|
||||
return Ok(None);
|
||||
}
|
||||
Ok(Some(serde_json::from_str::<CachedPost>(&payload).map_err(
|
||||
|e| rusqlite::Error::ToSqlConversionFailure(Box::new(e)),
|
||||
)?))
|
||||
})
|
||||
.await;
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(move |conn| {
|
||||
let mut stmt =
|
||||
conn.prepare("SELECT payload, created_at FROM link_cache WHERE url = ?1")?;
|
||||
let mut rows = stmt.query(params![key])?;
|
||||
let Some(row) = rows.next()? else {
|
||||
return Ok(None);
|
||||
};
|
||||
let payload: String = row.get(0)?;
|
||||
let created_at: f64 = row.get(1)?;
|
||||
if now_f64() - created_at > ttl {
|
||||
conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key])?;
|
||||
return Ok(None);
|
||||
}
|
||||
match serde_json::from_str::<CachedPost>(&payload) {
|
||||
Ok(post) => Ok(Some(post)),
|
||||
Err(e) => {
|
||||
// Unreadable payload (e.g. an older schema): drop it
|
||||
// instead of re-failing the parse on every later hit.
|
||||
conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key])?;
|
||||
Err(rusqlite::Error::ToSqlConversionFailure(Box::new(e)))
|
||||
}
|
||||
}
|
||||
})
|
||||
.await;
|
||||
match result {
|
||||
Ok(v) => v,
|
||||
Err(e) => {
|
||||
@@ -99,14 +105,16 @@ impl LinkCache {
|
||||
pub async fn put(&self, key: &str, post: &CachedPost) {
|
||||
let key = key.to_string();
|
||||
let payload = serde_json::to_string(post).expect("cached post serializes");
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
conn.execute(
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(move |conn| {
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO link_cache (url, payload, created_at) VALUES (?1, ?2, ?3)",
|
||||
params![key, payload, now_f64()],
|
||||
)?;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
if let Err(e) = result {
|
||||
log::error!("link cache write failed: {e}");
|
||||
}
|
||||
@@ -115,11 +123,13 @@ impl LinkCache {
|
||||
/// Drops an entry (e.g. a cached file id that turned out invalid).
|
||||
pub async fn remove(&self, key: &str) {
|
||||
let key = key.to_string();
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key])?;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(move |conn| {
|
||||
conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key])?;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
if let Err(e) = result {
|
||||
log::error!("link cache delete failed: {e}");
|
||||
}
|
||||
@@ -128,13 +138,15 @@ impl LinkCache {
|
||||
/// Removes expired entries; returns how many were deleted.
|
||||
pub async fn prune(&self, ttl: Duration) -> usize {
|
||||
let cutoff = now_f64() - ttl.as_secs_f64();
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
conn.execute(
|
||||
"DELETE FROM link_cache WHERE created_at < ?1",
|
||||
params![cutoff],
|
||||
)
|
||||
})
|
||||
.await;
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(move |conn| {
|
||||
conn.execute(
|
||||
"DELETE FROM link_cache WHERE created_at < ?1",
|
||||
params![cutoff],
|
||||
)
|
||||
})
|
||||
.await;
|
||||
match result {
|
||||
Ok(n) => n,
|
||||
Err(e) => {
|
||||
@@ -148,11 +160,13 @@ impl LinkCache {
|
||||
/// `key` is `None`. Returns how many rows were removed.
|
||||
pub async fn clear(&self, key: Option<&str>) -> usize {
|
||||
let key = key.map(str::to_string);
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| match &key {
|
||||
Some(key) => conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key]),
|
||||
None => conn.execute("DELETE FROM link_cache", []),
|
||||
})
|
||||
.await;
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(move |conn| match &key {
|
||||
Some(key) => conn.execute("DELETE FROM link_cache WHERE url = ?1", params![key]),
|
||||
None => conn.execute("DELETE FROM link_cache", []),
|
||||
})
|
||||
.await;
|
||||
match result {
|
||||
Ok(n) => n,
|
||||
Err(e) => {
|
||||
@@ -163,13 +177,6 @@ impl LinkCache {
|
||||
}
|
||||
}
|
||||
|
||||
fn now_f64() -> f64 {
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_secs_f64())
|
||||
.unwrap_or(0.0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -179,6 +186,7 @@ mod tests {
|
||||
url: "https://x.com/u/status/1".into(),
|
||||
caption: "cap".into(),
|
||||
title: "t".into(),
|
||||
content: "c".into(),
|
||||
author: "a".into(),
|
||||
author_url: "au".into(),
|
||||
tags: "".into(),
|
||||
@@ -190,10 +198,55 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// A payload written before the title/content split has no `content`
|
||||
/// field. It must still read back — the cache deletes what it cannot
|
||||
/// parse — with its text left where it was stored (`title`) and the
|
||||
/// caption it replays untouched. No migration: a self-hosted cache entry
|
||||
/// lives one TTL, and moving the text would only reshuffle `/set_format`
|
||||
/// placeholders until it expires.
|
||||
#[tokio::test]
|
||||
async fn pre_split_entry_still_parses() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let cache = LinkCache::new(
|
||||
crate::db::open_store(dir.path().join("c.db").to_str().unwrap()).unwrap(),
|
||||
);
|
||||
let legacy = serde_json::json!({
|
||||
"url": "https://x.com/u/status/1",
|
||||
"caption": "https://x.com/u/status/1\n<a href=\"au\">a</a>: old text",
|
||||
"title": "old text",
|
||||
"author": "a",
|
||||
"author_url": "au",
|
||||
"tags": "",
|
||||
"sensitive": false,
|
||||
"media": [{"kind": "photo", "file_id": "AgAC..."}]
|
||||
});
|
||||
{
|
||||
let conn = rusqlite::Connection::open(dir.path().join("c.db")).unwrap();
|
||||
conn.execute(
|
||||
"INSERT INTO link_cache (url, payload, created_at) VALUES (?1, ?2, ?3)",
|
||||
params!["twitter:1", legacy.to_string(), now_f64()],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let got = cache
|
||||
.get("twitter:1", Duration::from_secs(3600))
|
||||
.await
|
||||
.expect("a pre-split payload must not be dropped");
|
||||
assert_eq!(got.title, "old text");
|
||||
assert_eq!(got.content, "");
|
||||
assert_eq!(
|
||||
got.caption,
|
||||
"https://x.com/u/status/1\n<a href=\"au\">a</a>: old text"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_get_roundtrip() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let cache = LinkCache::open(dir.path().join("c.db").to_str().unwrap());
|
||||
let cache = LinkCache::new(
|
||||
crate::db::open_store(dir.path().join("c.db").to_str().unwrap()).unwrap(),
|
||||
);
|
||||
cache.put("twitter:1", &entry()).await;
|
||||
let got = cache.get("twitter:1", Duration::from_secs(3600)).await;
|
||||
assert!(got.is_some());
|
||||
@@ -205,11 +258,13 @@ mod tests {
|
||||
#[tokio::test]
|
||||
async fn expired_entry_removed_on_read() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let cache = LinkCache::open(dir.path().join("c.db").to_str().unwrap());
|
||||
let cache = LinkCache::new(
|
||||
crate::db::open_store(dir.path().join("c.db").to_str().unwrap()).unwrap(),
|
||||
);
|
||||
cache.put("twitter:1", &entry()).await;
|
||||
// Force the row into the past so a 1s TTL expires it.
|
||||
{
|
||||
let conn = Connection::open(dir.path().join("c.db")).unwrap();
|
||||
let conn = rusqlite::Connection::open(dir.path().join("c.db")).unwrap();
|
||||
conn.execute("UPDATE link_cache SET created_at = created_at - 100", [])
|
||||
.unwrap();
|
||||
}
|
||||
@@ -227,10 +282,38 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn unreadable_entry_is_dropped_on_read() {
|
||||
// A payload from an older schema must not be re-parsed on every hit:
|
||||
// the row is removed and the read reports a miss.
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let db_path = dir.path().join("c.db");
|
||||
let cache = LinkCache::new(crate::db::open_store(db_path.to_str().unwrap()).unwrap());
|
||||
{
|
||||
let conn = rusqlite::Connection::open(&db_path).unwrap();
|
||||
conn.execute(
|
||||
"INSERT INTO link_cache (url, payload, created_at) VALUES (?1, ?2, ?3)",
|
||||
params!["twitter:1", "{not json", now_f64()],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
assert!(
|
||||
cache
|
||||
.get("twitter:1", Duration::from_secs(3600))
|
||||
.await
|
||||
.is_none()
|
||||
);
|
||||
// Dropped, not left behind for the next hit.
|
||||
assert_eq!(cache.clear(None).await, 0, "corrupted row still present");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn remove_and_prune() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let cache = LinkCache::open(dir.path().join("c.db").to_str().unwrap());
|
||||
let cache = LinkCache::new(
|
||||
crate::db::open_store(dir.path().join("c.db").to_str().unwrap()).unwrap(),
|
||||
);
|
||||
cache.put("twitter:1", &entry()).await;
|
||||
cache.put("pixiv:2", &entry()).await;
|
||||
cache.remove("twitter:1").await;
|
||||
@@ -247,7 +330,7 @@ mod tests {
|
||||
.is_some()
|
||||
);
|
||||
{
|
||||
let conn = Connection::open(dir.path().join("c.db")).unwrap();
|
||||
let conn = rusqlite::Connection::open(dir.path().join("c.db")).unwrap();
|
||||
conn.execute("UPDATE link_cache SET created_at = created_at - 100", [])
|
||||
.unwrap();
|
||||
}
|
||||
@@ -263,7 +346,9 @@ mod tests {
|
||||
#[tokio::test]
|
||||
async fn clear_one_entry_or_all() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let cache = LinkCache::open(dir.path().join("c.db").to_str().unwrap());
|
||||
let cache = LinkCache::new(
|
||||
crate::db::open_store(dir.path().join("c.db").to_str().unwrap()).unwrap(),
|
||||
);
|
||||
cache.put("twitter:1", &entry()).await;
|
||||
cache.put("pixiv:2", &entry()).await;
|
||||
// By key: only the matching row is removed.
|
||||
|
||||
@@ -8,14 +8,18 @@ use tokio::sync::watch;
|
||||
use x_media::site;
|
||||
|
||||
mod config;
|
||||
mod ctx;
|
||||
mod db;
|
||||
mod handlers;
|
||||
mod link_cache;
|
||||
mod media_sender;
|
||||
mod photo;
|
||||
mod queue;
|
||||
mod rate_limit;
|
||||
mod send;
|
||||
mod state;
|
||||
|
||||
use ctx::CONTEXT;
|
||||
use handlers::{CHAT_STORE, CONFIG, LINK_CACHE, TASK_QUEUE};
|
||||
|
||||
/// Docker `stop` / `compose down` delivers SIGTERM, which teloxide's ctrlc
|
||||
@@ -43,6 +47,14 @@ async fn main() {
|
||||
log::info!("Starting bot");
|
||||
|
||||
let bot = Bot::from_env();
|
||||
// Force the queue workers' shared Bot to initialize now so a missing
|
||||
// token fails at startup, not on the first queued task.
|
||||
let _ = &*send::BOT;
|
||||
|
||||
// Register the command list with Telegram (client `/` menu).
|
||||
if let Err(e) = handlers::register_commands(&bot).await {
|
||||
log::warn!("failed to register commands: {e}");
|
||||
}
|
||||
|
||||
log::info!(
|
||||
"config: {} admin(s), edit-message TTL {}s",
|
||||
@@ -51,25 +63,32 @@ async fn main() {
|
||||
);
|
||||
|
||||
// Queue worker: handles typed tasks, dead-letters failed sends to the
|
||||
// task's chat.
|
||||
// task's chat. Both closures use the shared context (the queue requires
|
||||
// 'static handlers, and the statics are process-wide anyway).
|
||||
TASK_QUEUE
|
||||
.start(send::handle_task, send::dead_letter_notify)
|
||||
.start(
|
||||
|payload| send::handle_task(&CONTEXT, payload),
|
||||
|payload, message| send::dead_letter_notify(&CONTEXT, payload, message),
|
||||
)
|
||||
.await;
|
||||
log::info!("task queue worker started");
|
||||
|
||||
// Pixiv login validation (user request): a failed login notifies the
|
||||
// admin and disables pixiv for this process.
|
||||
if site::pixiv::enabled() {
|
||||
match site::pixiv::validate().await {
|
||||
Ok(()) => log::info!("pixiv login validated"),
|
||||
Err(e) => {
|
||||
log::error!("pixiv login failed: {e}");
|
||||
if let Some(admin) = CONFIG.admin_ids.first() {
|
||||
let _ = bot
|
||||
.send_message(ChatId(*admin), format!("Pixiv login failed: {e}"))
|
||||
.await;
|
||||
}
|
||||
site::pixiv::disable();
|
||||
// URL job workers: bounded channel + fixed pool for per-URL work.
|
||||
handlers::start_url_workers().await;
|
||||
log::info!("url workers started");
|
||||
|
||||
// Site login validation (user request): a failed login notifies the
|
||||
// admin and the site disables itself for this process (pixiv).
|
||||
let failures = site::validate_all().await;
|
||||
if failures.is_empty() {
|
||||
log::info!("site logins validated");
|
||||
} else {
|
||||
for (site_id, message) in &failures {
|
||||
log::error!("{site_id} login failed: {message}");
|
||||
if let Some(admin) = CONFIG.admin_ids.first() {
|
||||
let _ = bot
|
||||
.send_message(ChatId(*admin), format!("{site_id} login failed: {message}"))
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -95,6 +114,10 @@ async fn main() {
|
||||
if pruned > 0 {
|
||||
log::info!("link cache: pruned {pruned} expired entr(ies)");
|
||||
}
|
||||
let idle_limiters = crate::rate_limit::prune_idle();
|
||||
if idle_limiters > 0 {
|
||||
log::debug!("rate limiter: dropped {idle_limiters} idle bucket(s)");
|
||||
}
|
||||
for (chat_id, prompt_message_id) in removed {
|
||||
// If the prompt was already deleted, this fails with a
|
||||
// 400 "message to edit not found" — log and ignore.
|
||||
@@ -166,12 +189,25 @@ async fn main() {
|
||||
.await;
|
||||
}
|
||||
|
||||
// Graceful stop (Ctrl+C / SIGTERM): stop the sweep, notify the admin, drain the queue.
|
||||
// Graceful stop (Ctrl+C / SIGTERM): stop the sweep, notify the admin,
|
||||
// drain the queue. Bounded: a worker mid-download (30 s timeout) or a
|
||||
// long ugoira encode must not hold the shutdown hostage forever.
|
||||
log::info!("Stopping bot");
|
||||
let _ = stop_tx.send(true);
|
||||
if let Some(admin) = CONFIG.admin_ids.first() {
|
||||
let _ = bot.send_message(ChatId(*admin), "Shutting down...").await;
|
||||
const SHUTDOWN_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(30);
|
||||
let shutdown = async {
|
||||
let _ = stop_tx.send(true);
|
||||
handlers::stop_url_workers().await;
|
||||
if let Some(admin) = CONFIG.admin_ids.first() {
|
||||
let _ = bot.send_message(ChatId(*admin), "Shutting down...").await;
|
||||
}
|
||||
TASK_QUEUE.stop().await;
|
||||
};
|
||||
if tokio::time::timeout(SHUTDOWN_TIMEOUT, shutdown)
|
||||
.await
|
||||
.is_err()
|
||||
{
|
||||
log::warn!("graceful shutdown timed out after {SHUTDOWN_TIMEOUT:?}; exiting");
|
||||
} else {
|
||||
log::info!("Bot stopped");
|
||||
}
|
||||
TASK_QUEUE.stop().await;
|
||||
log::info!("Bot stopped");
|
||||
}
|
||||
|
||||
@@ -0,0 +1,453 @@
|
||||
//! Send abstraction: the message-sending surface [`send`](crate::send)
|
||||
//! needs, so the send pipeline can be tested with a scripted mock instead of
|
||||
//! a live teloxide `Bot`.
|
||||
|
||||
use std::future::Future;
|
||||
use std::pin::Pin;
|
||||
use teloxide::RequestError;
|
||||
use teloxide::prelude::Requester;
|
||||
use teloxide::prelude::*;
|
||||
use teloxide::types::{
|
||||
CallbackQueryId, ChatAction, ChatId, InlineKeyboardMarkup, InputFile, InputMedia, Message,
|
||||
MessageId, ParseMode, ReplyParameters,
|
||||
};
|
||||
|
||||
/// Boxed, `Send` future returned by a [`MediaSender`] method (`async fn` in
|
||||
/// traits is not dyn-compatible).
|
||||
type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + Send + 'a>>;
|
||||
|
||||
/// The message-sending surface the send pipeline uses. The production
|
||||
/// implementation is teloxide's [`Bot`]; tests inject a scripted mock to
|
||||
/// cover the fallback and classification logic without touching the
|
||||
/// Telegram API.
|
||||
pub trait MediaSender: Send + Sync {
|
||||
/// Sends a media group, replying to `reply_to`.
|
||||
fn send_media_group(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
reply_to: MessageId,
|
||||
items: Vec<InputMedia>,
|
||||
) -> BoxFuture<'_, Result<Vec<Message>, RequestError>>;
|
||||
|
||||
/// Sends a lone animation, replying to `reply_to`.
|
||||
fn send_animation<'a>(
|
||||
&'a self,
|
||||
chat_id: ChatId,
|
||||
reply_to: MessageId,
|
||||
caption: &'a str,
|
||||
spoiler: bool,
|
||||
file: InputFile,
|
||||
) -> BoxFuture<'a, Result<Message, RequestError>>;
|
||||
|
||||
/// Copies messages between chats (forward to channel).
|
||||
fn copy_messages(
|
||||
&self,
|
||||
to: ChatId,
|
||||
from: ChatId,
|
||||
ids: Vec<MessageId>,
|
||||
) -> BoxFuture<'_, Result<Vec<MessageId>, RequestError>>;
|
||||
|
||||
/// Sends a plain text message, optionally replying to `reply_to` and
|
||||
/// attaching `reply_markup`. Returns the sent message's id: the bot only
|
||||
/// ever needs that (the edit-before-forward prompt's record is keyed by
|
||||
/// it), and returning the whole `Message` would force every test mock to
|
||||
/// construct one.
|
||||
fn send_message(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
text: String,
|
||||
reply_to: Option<MessageId>,
|
||||
reply_markup: Option<InlineKeyboardMarkup>,
|
||||
) -> BoxFuture<'_, Result<i64, RequestError>>;
|
||||
|
||||
/// Answers a callback query, optionally with a toast `text` shown to the
|
||||
/// user who pressed the button.
|
||||
fn answer_callback_query(
|
||||
&self,
|
||||
id: CallbackQueryId,
|
||||
text: Option<String>,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>>;
|
||||
|
||||
/// Rewrites a message's caption, always with HTML parse mode (every caller
|
||||
/// in this bot renders escaped HTML: templates and edit-before-forward
|
||||
/// links).
|
||||
fn edit_message_caption(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
message_id: MessageId,
|
||||
caption: String,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>>;
|
||||
|
||||
/// Deletes a message (the edit-before-forward prompt after a forward).
|
||||
fn delete_message(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
message_id: MessageId,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>>;
|
||||
|
||||
/// Sets the chat's "typing / uploading …" indicator (cosmetic).
|
||||
fn send_chat_action(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
action: ChatAction,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>>;
|
||||
}
|
||||
|
||||
impl MediaSender for Bot {
|
||||
fn send_media_group(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
reply_to: MessageId,
|
||||
items: Vec<InputMedia>,
|
||||
) -> BoxFuture<'_, Result<Vec<Message>, RequestError>> {
|
||||
Box::pin(async move {
|
||||
// Pace media sends per chat (one token per item) so bursts do not
|
||||
// trip Telegram's flood control.
|
||||
crate::rate_limit::limiter_for(chat_id.0)
|
||||
.acquire(items.len() as f64)
|
||||
.await;
|
||||
// `<Bot as Requester>::` disambiguates from this trait's same-named
|
||||
// method (teloxide's API lives in the `Requester` trait).
|
||||
<Bot as Requester>::send_media_group(self, chat_id, items)
|
||||
.reply_parameters(ReplyParameters::new(reply_to).allow_sending_without_reply())
|
||||
.await
|
||||
})
|
||||
}
|
||||
|
||||
fn send_animation<'a>(
|
||||
&'a self,
|
||||
chat_id: ChatId,
|
||||
reply_to: MessageId,
|
||||
caption: &'a str,
|
||||
spoiler: bool,
|
||||
file: InputFile,
|
||||
) -> BoxFuture<'a, Result<Message, RequestError>> {
|
||||
Box::pin(async move {
|
||||
crate::rate_limit::limiter_for(chat_id.0).acquire(1.0).await;
|
||||
let mut request = <Bot as Requester>::send_animation(self, chat_id, file)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html)
|
||||
.reply_parameters(ReplyParameters::new(reply_to).allow_sending_without_reply());
|
||||
if spoiler {
|
||||
request = request.has_spoiler(true);
|
||||
}
|
||||
request.await
|
||||
})
|
||||
}
|
||||
|
||||
fn copy_messages(
|
||||
&self,
|
||||
to: ChatId,
|
||||
from: ChatId,
|
||||
ids: Vec<MessageId>,
|
||||
) -> BoxFuture<'_, Result<Vec<MessageId>, RequestError>> {
|
||||
Box::pin(async move {
|
||||
// Channel forwards are the burstiest path (batch copies); pace
|
||||
// them per message against the channel's budget.
|
||||
crate::rate_limit::limiter_for(to.0)
|
||||
.acquire(ids.len() as f64)
|
||||
.await;
|
||||
<Bot as Requester>::copy_messages(self, to, from, ids).await
|
||||
})
|
||||
}
|
||||
|
||||
fn send_message(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
text: String,
|
||||
reply_to: Option<MessageId>,
|
||||
reply_markup: Option<InlineKeyboardMarkup>,
|
||||
) -> BoxFuture<'_, Result<i64, RequestError>> {
|
||||
Box::pin(async move {
|
||||
let mut request = <Bot as Requester>::send_message(self, chat_id, text);
|
||||
if let Some(reply_to) = reply_to {
|
||||
request = request
|
||||
.reply_parameters(ReplyParameters::new(reply_to).allow_sending_without_reply());
|
||||
}
|
||||
if let Some(markup) = reply_markup {
|
||||
request = request.reply_markup(markup);
|
||||
}
|
||||
request.await.map(|message| message.id.0 as i64)
|
||||
})
|
||||
}
|
||||
|
||||
fn answer_callback_query(
|
||||
&self,
|
||||
id: CallbackQueryId,
|
||||
text: Option<String>,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>> {
|
||||
Box::pin(async move {
|
||||
let mut request = <Bot as Requester>::answer_callback_query(self, id);
|
||||
if let Some(text) = text {
|
||||
request = request.text(text);
|
||||
}
|
||||
request.await.map(|_| ())
|
||||
})
|
||||
}
|
||||
|
||||
fn edit_message_caption(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
message_id: MessageId,
|
||||
caption: String,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>> {
|
||||
Box::pin(async move {
|
||||
<Bot as Requester>::edit_message_caption(self, chat_id, message_id)
|
||||
.caption(caption)
|
||||
.parse_mode(ParseMode::Html)
|
||||
.await
|
||||
.map(|_| ())
|
||||
})
|
||||
}
|
||||
|
||||
fn delete_message(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
message_id: MessageId,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>> {
|
||||
Box::pin(async move {
|
||||
<Bot as Requester>::delete_message(self, chat_id, message_id)
|
||||
.await
|
||||
.map(|_| ())
|
||||
})
|
||||
}
|
||||
|
||||
fn send_chat_action(
|
||||
&self,
|
||||
chat_id: ChatId,
|
||||
action: ChatAction,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>> {
|
||||
Box::pin(async move {
|
||||
// teloxide's `send_chat_action` returns `Result<True, _>` (its
|
||||
// unit marker type); map the success to `()`.
|
||||
<Bot as Requester>::send_chat_action(self, chat_id, action)
|
||||
.await
|
||||
.map(|_| ())
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Test support: a scripted [`MediaSender`] mock (no Telegram API involved).
|
||||
#[cfg(test)]
|
||||
pub(crate) mod test_support {
|
||||
use super::*;
|
||||
use parking_lot::Mutex;
|
||||
|
||||
/// One scripted outcome, consumed front-to-back; the last entry repeats
|
||||
/// for further calls of the same method kind.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub(crate) enum Outcome {
|
||||
GroupOk,
|
||||
GroupErr,
|
||||
AnimationErr,
|
||||
CopyOk,
|
||||
CopyErr,
|
||||
/// An error from `send_message` (replies are fire-and-forget, so an
|
||||
/// error is fine for tests).
|
||||
MessageErr,
|
||||
/// A successful `send_message`, returning message id [`MockSender::SENT_ID`].
|
||||
MessageOk,
|
||||
EditOk,
|
||||
EditErr,
|
||||
}
|
||||
|
||||
/// Replays a script and records what was sent, so tests can assert the
|
||||
/// user-visible text a path produced.
|
||||
pub(crate) struct MockSender {
|
||||
script: Mutex<Vec<Outcome>>,
|
||||
cursor: Mutex<usize>,
|
||||
calls: Mutex<Vec<&'static str>>,
|
||||
messages: Mutex<Vec<String>>,
|
||||
captions: Mutex<Vec<String>>,
|
||||
answers: Mutex<Vec<Option<String>>>,
|
||||
/// Builds the error every `*Err` outcome returns (RequestError is not
|
||||
/// cloneable, so the factory recreates it per call).
|
||||
error: Box<dyn Fn() -> RequestError + Send + Sync>,
|
||||
}
|
||||
|
||||
impl MockSender {
|
||||
/// The message id a successful `send_message` reports.
|
||||
pub(crate) const SENT_ID: i64 = 1;
|
||||
|
||||
pub(crate) fn scripted(
|
||||
script: Vec<Outcome>,
|
||||
error: impl Fn() -> RequestError + Send + Sync + 'static,
|
||||
) -> Self {
|
||||
MockSender {
|
||||
script: Mutex::new(script),
|
||||
cursor: Mutex::new(0),
|
||||
calls: Mutex::new(Vec::new()),
|
||||
messages: Mutex::new(Vec::new()),
|
||||
captions: Mutex::new(Vec::new()),
|
||||
answers: Mutex::new(Vec::new()),
|
||||
error: Box::new(error),
|
||||
}
|
||||
}
|
||||
|
||||
/// Method names in call order (e.g. `["send_media_group",
|
||||
/// "send_media_group"]` proves the fallback re-sent).
|
||||
pub(crate) fn calls(&self) -> Vec<&'static str> {
|
||||
self.calls.lock().clone()
|
||||
}
|
||||
|
||||
/// Texts of the plain messages sent, in order.
|
||||
pub(crate) fn messages(&self) -> Vec<String> {
|
||||
self.messages.lock().clone()
|
||||
}
|
||||
|
||||
/// Captions passed to `edit_message_caption`, in order.
|
||||
pub(crate) fn captions(&self) -> Vec<String> {
|
||||
self.captions.lock().clone()
|
||||
}
|
||||
|
||||
/// Toast texts of the answered callback queries, in order.
|
||||
pub(crate) fn answers(&self) -> Vec<Option<String>> {
|
||||
self.answers.lock().clone()
|
||||
}
|
||||
|
||||
fn next(&self, kind: &'static str) -> Outcome {
|
||||
self.calls.lock().push(kind);
|
||||
let script = self.script.lock();
|
||||
let mut cursor = self.cursor.lock();
|
||||
if script.is_empty() {
|
||||
panic!("mock script exhausted: {kind}");
|
||||
}
|
||||
let idx = (*cursor).min(script.len() - 1);
|
||||
*cursor = idx + 1;
|
||||
script[idx]
|
||||
}
|
||||
|
||||
fn error(&self) -> RequestError {
|
||||
(self.error)()
|
||||
}
|
||||
}
|
||||
|
||||
impl MediaSender for MockSender {
|
||||
fn send_media_group(
|
||||
&self,
|
||||
_chat_id: ChatId,
|
||||
_reply_to: MessageId,
|
||||
items: Vec<InputMedia>,
|
||||
) -> BoxFuture<'_, Result<Vec<Message>, RequestError>> {
|
||||
Box::pin(async move {
|
||||
// Record the captions exactly as Telegram receives them (only
|
||||
// the first item of a group carries one), so tests can assert
|
||||
// what a recipient sees.
|
||||
self.captions
|
||||
.lock()
|
||||
.extend(items.iter().filter_map(|item| match item {
|
||||
InputMedia::Photo(photo) => photo.caption.clone(),
|
||||
InputMedia::Video(video) => video.caption.clone(),
|
||||
InputMedia::Animation(animation) => animation.caption.clone(),
|
||||
_ => None,
|
||||
}));
|
||||
match self.next("send_media_group") {
|
||||
Outcome::GroupOk => Ok(Vec::new()),
|
||||
Outcome::GroupErr => Err(self.error()),
|
||||
other => panic!("unexpected outcome {other:?} for send_media_group"),
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn send_animation<'a>(
|
||||
&'a self,
|
||||
_chat_id: ChatId,
|
||||
_reply_to: MessageId,
|
||||
_caption: &'a str,
|
||||
_spoiler: bool,
|
||||
_file: InputFile,
|
||||
) -> BoxFuture<'a, Result<Message, RequestError>> {
|
||||
Box::pin(async move {
|
||||
match self.next("send_animation") {
|
||||
Outcome::AnimationErr => Err(self.error()),
|
||||
other => panic!("unexpected outcome {other:?} for send_animation"),
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn copy_messages(
|
||||
&self,
|
||||
_to: ChatId,
|
||||
_from: ChatId,
|
||||
_ids: Vec<MessageId>,
|
||||
) -> BoxFuture<'_, Result<Vec<MessageId>, RequestError>> {
|
||||
Box::pin(async move {
|
||||
match self.next("copy_messages") {
|
||||
Outcome::CopyOk => Ok(vec![MessageId(1)]),
|
||||
Outcome::CopyErr => Err(self.error()),
|
||||
other => panic!("unexpected outcome {other:?} for copy_messages"),
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn send_message(
|
||||
&self,
|
||||
_chat_id: ChatId,
|
||||
text: String,
|
||||
_reply_to: Option<MessageId>,
|
||||
_reply_markup: Option<InlineKeyboardMarkup>,
|
||||
) -> BoxFuture<'_, Result<i64, RequestError>> {
|
||||
Box::pin(async move {
|
||||
self.messages.lock().push(text);
|
||||
match self.next("send_message") {
|
||||
Outcome::MessageOk => Ok(MockSender::SENT_ID),
|
||||
Outcome::MessageErr => Err(self.error()),
|
||||
other => panic!("unexpected outcome {other:?} for send_message"),
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn answer_callback_query(
|
||||
&self,
|
||||
_id: CallbackQueryId,
|
||||
text: Option<String>,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>> {
|
||||
// Always succeeds: the toast is cosmetic, so the script stays
|
||||
// focused on the outcomes a test cares about.
|
||||
Box::pin(async move {
|
||||
self.calls.lock().push("answer_callback_query");
|
||||
self.answers.lock().push(text);
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
|
||||
fn edit_message_caption(
|
||||
&self,
|
||||
_chat_id: ChatId,
|
||||
_message_id: MessageId,
|
||||
caption: String,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>> {
|
||||
Box::pin(async move {
|
||||
self.captions.lock().push(caption);
|
||||
match self.next("edit_message_caption") {
|
||||
Outcome::EditOk => Ok(()),
|
||||
Outcome::EditErr => Err(self.error()),
|
||||
other => panic!("unexpected outcome {other:?} for edit_message_caption"),
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn delete_message(
|
||||
&self,
|
||||
_chat_id: ChatId,
|
||||
_message_id: MessageId,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>> {
|
||||
// Deletion is fire-and-forget in every caller; always succeeds.
|
||||
Box::pin(async move {
|
||||
self.calls.lock().push("delete_message");
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
|
||||
fn send_chat_action(
|
||||
&self,
|
||||
_chat_id: ChatId,
|
||||
_action: ChatAction,
|
||||
) -> BoxFuture<'_, Result<(), RequestError>> {
|
||||
Box::pin(async move {
|
||||
self.calls.lock().push("send_chat_action");
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -26,8 +26,9 @@ pub const PHOTO_TARGET_DIMENSION_SUM: u32 = 9900;
|
||||
/// to a smaller media URL instead.
|
||||
pub const MAX_UPLOAD_BYTES: u64 = 10 * 1024 * 1024;
|
||||
/// Decode budget (bytes): a larger intermediate buffer is not worth the peak
|
||||
/// memory; the photo degrades to the smaller URL instead.
|
||||
const MAX_DECODE_BYTES: u64 = 512 * 1024 * 1024;
|
||||
/// memory; the photo degrades to the smaller URL instead. Also the cap for
|
||||
/// downloading photos in the send fallback (they must be downloaded whole).
|
||||
pub(crate) const MAX_DECODE_BYTES: u64 = 512 * 1024 * 1024;
|
||||
/// JPEG output quality (1-100).
|
||||
const JPEG_QUALITY: u8 = 90;
|
||||
|
||||
@@ -66,8 +67,9 @@ impl PixBuf {
|
||||
}
|
||||
|
||||
/// Entry point: detects the format and processes the photo if needed.
|
||||
pub fn prepare_photo(file: NamedTempFile) -> Result<PhotoPrep, String> {
|
||||
let bytes = std::fs::read(file.path()).map_err(|e| format!("prepare read failed: {e}"))?;
|
||||
/// The caller hands in the already-downloaded bytes (they are in memory from
|
||||
/// the download anyway; re-reading the temp file would double the I/O).
|
||||
pub fn prepare_photo(file: NamedTempFile, bytes: &[u8]) -> Result<PhotoPrep, String> {
|
||||
if bytes.starts_with(b"\x89PNG\r\n\x1a\n") {
|
||||
prepare_png(file, bytes)
|
||||
} else if bytes.starts_with(&[0xFF, 0xD8, 0xFF]) {
|
||||
@@ -120,7 +122,7 @@ fn output_channels(color: png::ColorType) -> usize {
|
||||
/// white; 16-bit per channel was already stripped to 8-bit at decode.
|
||||
fn flatten_rgba_to_rgb(rgba: &[u8]) -> Vec<u8> {
|
||||
let mut rgb = Vec::with_capacity(rgba.len() / 4 * 3);
|
||||
for px in rgba.chunks_exact(4) {
|
||||
for px in rgba.as_chunks::<4>().0 {
|
||||
let a = px[3] as u32;
|
||||
for v in &px[..3] {
|
||||
// Over white: C = C*a/255 + 255*(1 - a/255).
|
||||
@@ -176,7 +178,9 @@ fn encode_jpeg(pix: &PixBuf, w: u32, h: u32) -> Result<Vec<u8>, String> {
|
||||
PixBuf::GrayAlpha(v) => {
|
||||
// JPEG has no alpha: composite onto white, output as gray.
|
||||
let gray: Vec<u8> = v
|
||||
.chunks_exact(2)
|
||||
.as_chunks::<2>()
|
||||
.0
|
||||
.iter()
|
||||
.map(|px| {
|
||||
let (g, a) = (px[0] as u32, px[1] as u32);
|
||||
((g * a + 255 * (255 - a)) / 255).min(255) as u8
|
||||
@@ -215,13 +219,13 @@ fn target_dims(w: u32, h: u32) -> (u32, u32) {
|
||||
/// PNG branch: decode (16→8, palette→RGB; gray/GA stay), flatten RGBA to
|
||||
/// RGB, Lanczos-downscale beyond the dimension cap, encode PNG — a PNG still
|
||||
/// over the upload cap afterwards becomes JPEG.
|
||||
fn prepare_png(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String> {
|
||||
let (w, h, _bit_depth, color_type) = parse_png_header(&bytes).ok_or("invalid PNG header")?;
|
||||
fn prepare_png(file: NamedTempFile, bytes: &[u8]) -> Result<PhotoPrep, String> {
|
||||
let (w, h, _bit_depth, color_type) = parse_png_header(bytes).ok_or("invalid PNG header")?;
|
||||
let size_over = bytes.len() as u64 > MAX_UPLOAD_BYTES;
|
||||
if w + h <= PHOTO_MAX_DIMENSION_SUM && !size_over {
|
||||
return Ok(PhotoPrep::Upload(file));
|
||||
}
|
||||
log::info!(
|
||||
log::debug!(
|
||||
"photo {w}x{h} ({_bit_depth:?} {color_type:?}, {} bytes) needs processing",
|
||||
bytes.len()
|
||||
);
|
||||
@@ -238,7 +242,7 @@ fn prepare_png(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String>
|
||||
png::ColorType::Indexed => png::Transformations::EXPAND,
|
||||
_ => png::Transformations::STRIP_16,
|
||||
};
|
||||
let mut decoder = png::Decoder::new(std::io::Cursor::new(&bytes));
|
||||
let mut decoder = png::Decoder::new(std::io::Cursor::new(bytes));
|
||||
decoder.set_transformations(transforms);
|
||||
let mut reader = decoder
|
||||
.read_info()
|
||||
@@ -267,7 +271,7 @@ fn prepare_png(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String>
|
||||
let (nw, nh) = target_dims(w, h);
|
||||
pix = resize_pix(pix, w, h, nw, nh)?;
|
||||
(w, h) = (nw, nh);
|
||||
log::info!("downscaled photo to {w}x{h} (Lanczos3)");
|
||||
log::debug!("downscaled photo to {w}x{h} (Lanczos3)");
|
||||
}
|
||||
|
||||
let mut png_bytes = Vec::new();
|
||||
@@ -275,7 +279,7 @@ fn prepare_png(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String>
|
||||
if png_bytes.len() as u64 <= MAX_UPLOAD_BYTES {
|
||||
return Ok(PhotoPrep::Upload(write_temp(&png_bytes, "png")?));
|
||||
}
|
||||
log::info!("PNG still over the upload cap after processing; transcoding to JPEG");
|
||||
log::debug!("PNG still over the upload cap after processing; transcoding to JPEG");
|
||||
let jpeg_bytes = encode_jpeg(&pix, w, h)?;
|
||||
if jpeg_bytes.len() as u64 <= MAX_UPLOAD_BYTES {
|
||||
return Ok(PhotoPrep::Upload(write_temp(&jpeg_bytes, "jpg")?));
|
||||
@@ -285,8 +289,8 @@ fn prepare_png(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String>
|
||||
}
|
||||
|
||||
/// JPEG branch: zune-jpeg decode → Lanczos downscale → jpeg-encoder output.
|
||||
fn prepare_jpeg(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String> {
|
||||
let mut decoder = zune_jpeg::JpegDecoder::new(std::io::Cursor::new(&bytes));
|
||||
fn prepare_jpeg(file: NamedTempFile, bytes: &[u8]) -> Result<PhotoPrep, String> {
|
||||
let mut decoder = zune_jpeg::JpegDecoder::new(std::io::Cursor::new(bytes));
|
||||
// Decodes to RGB by default. Headers first so dimensions are known before
|
||||
// the (potentially huge) pixel decode.
|
||||
decoder
|
||||
@@ -309,7 +313,7 @@ fn prepare_jpeg(file: NamedTempFile, bytes: Vec<u8>) -> Result<PhotoPrep, String
|
||||
let (nw, nh) = target_dims(w, h);
|
||||
pix = resize_pix(pix, w, h, nw, nh)?;
|
||||
(w, h) = (nw, nh);
|
||||
log::info!("downscaled jpeg to {w}x{h} (Lanczos3)");
|
||||
log::debug!("downscaled jpeg to {w}x{h} (Lanczos3)");
|
||||
}
|
||||
let jpeg_bytes = encode_jpeg(&pix, w, h)?;
|
||||
if jpeg_bytes.len() as u64 <= MAX_UPLOAD_BYTES {
|
||||
@@ -422,7 +426,7 @@ mod tests {
|
||||
|
||||
let mut file = tempfile::Builder::new().suffix(".png").tempfile().unwrap();
|
||||
std::io::Write::write_all(file.as_file_mut(), &bytes).unwrap();
|
||||
prepare_photo(file)
|
||||
prepare_photo(file, &bytes)
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -467,7 +471,7 @@ mod tests {
|
||||
}
|
||||
let mut file = tempfile::Builder::new().suffix(".jpg").tempfile().unwrap();
|
||||
std::io::Write::write_all(file.as_file_mut(), &bytes).unwrap();
|
||||
match prepare_photo(file).unwrap() {
|
||||
match prepare_photo(file, &bytes).unwrap() {
|
||||
PhotoPrep::Upload(file) => {
|
||||
let out = std::fs::read(file.path()).unwrap();
|
||||
assert!(out.starts_with(&[0xFF, 0xD8]), "output must stay jpeg");
|
||||
@@ -515,7 +519,7 @@ mod tests {
|
||||
|
||||
let mut file = tempfile::Builder::new().suffix(".png").tempfile().unwrap();
|
||||
std::io::Write::write_all(file.as_file_mut(), &bytes).unwrap();
|
||||
match prepare_photo(file).unwrap() {
|
||||
match prepare_photo(file, &bytes).unwrap() {
|
||||
PhotoPrep::Upload(file) => {
|
||||
let out = std::fs::read(file.path()).unwrap();
|
||||
assert!(out.starts_with(&[0xFF, 0xD8]), "must transcode to JPEG");
|
||||
|
||||
+222
-87
@@ -5,13 +5,14 @@
|
||||
//! flow. The Python dict-mutation hack (attempts inside the payload) is
|
||||
//! replaced by dedicated columns.
|
||||
|
||||
use crate::db::now_f64;
|
||||
use parking_lot::Mutex;
|
||||
use rusqlite::{Connection, TransactionBehavior, params};
|
||||
use rusqlite::{TransactionBehavior, params};
|
||||
use serde_json::Value;
|
||||
use std::pin::Pin;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::time::{Duration, SystemTime, UNIX_EPOCH};
|
||||
use std::time::Duration;
|
||||
use tokio::sync::Notify;
|
||||
use tokio::task::JoinHandle;
|
||||
|
||||
@@ -39,8 +40,14 @@ type Handler = dyn Fn(Value) -> BoxFuture<'static, Result<(), QueueError>> + Sen
|
||||
type DeadLetter = dyn Fn(Value, String) -> BoxFuture<'static, ()> + Send + Sync;
|
||||
|
||||
pub struct PersistentTaskQueue {
|
||||
db_path: String,
|
||||
pool: std::sync::Arc<crate::db::DbPool>,
|
||||
/// Wakes the workers when a row becomes leasable. `notify_one` stores a
|
||||
/// permit, so nothing else may share it: a waiter that is not a worker
|
||||
/// (the sweep) can consume the permit and leave the due row pending until
|
||||
/// the next enqueue.
|
||||
notify: Arc<Notify>,
|
||||
/// Wakes the lease-expiry sweep; `stop` is the only producer.
|
||||
sweep_notify: Arc<Notify>,
|
||||
stop: Arc<AtomicBool>,
|
||||
worker: Mutex<Vec<JoinHandle<()>>>,
|
||||
counter: AtomicU64,
|
||||
@@ -53,48 +60,41 @@ struct LeasedRow {
|
||||
}
|
||||
|
||||
/// Owned worker state so the spawned loop does not borrow the queue handle.
|
||||
#[derive(Clone)]
|
||||
struct QueueWorker {
|
||||
db_path: String,
|
||||
pool: std::sync::Arc<crate::db::DbPool>,
|
||||
notify: Arc<Notify>,
|
||||
stop: Arc<AtomicBool>,
|
||||
handler: Arc<Handler>,
|
||||
dead_letter: Arc<DeadLetter>,
|
||||
}
|
||||
|
||||
fn now_f64() -> f64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.map(|d| d.as_secs_f64())
|
||||
.unwrap_or(0.0)
|
||||
/// Resets rows left `in_progress` with an expired lock TTL back to `pending`
|
||||
/// so they can be leased again (crash/panic recovery).
|
||||
fn recover_update(conn: &rusqlite::Connection) -> rusqlite::Result<()> {
|
||||
conn.execute(
|
||||
"UPDATE tasks SET status='pending', locked_until=0 WHERE status='in_progress' AND locked_until < ?1",
|
||||
params![now_f64()],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn ensure_schema(conn: &rusqlite::Connection) -> rusqlite::Result<()> {
|
||||
conn.execute_batch(
|
||||
"CREATE TABLE IF NOT EXISTS tasks (id TEXT PRIMARY KEY, payload TEXT NOT NULL, \
|
||||
run_after REAL NOT NULL, attempts INTEGER NOT NULL, status TEXT NOT NULL, \
|
||||
locked_until REAL NOT NULL, created_at REAL NOT NULL);",
|
||||
)
|
||||
/// Base delay × 2^attempts (attempts = retries already done), capped at 300s.
|
||||
/// Applied at the queue layer so the attempt count actually reaches the
|
||||
/// backoff computation; Telegram `RetryAfter` delays get the same treatment
|
||||
/// (conservatively larger wait, no API change needed).
|
||||
fn scaled_retry_delay(base: f64, attempts: i32) -> f64 {
|
||||
(base * 2f64.powi(attempts)).min(300.0)
|
||||
}
|
||||
|
||||
impl PersistentTaskQueue {
|
||||
pub fn new(db_path: &str) -> Self {
|
||||
// Ensure the parent dir and table exist even if only the queue (not
|
||||
// ChatStore) is used — a fresh container without a mounted data dir
|
||||
// must still be able to open the DB.
|
||||
if let Some(parent) = std::path::Path::new(db_path).parent()
|
||||
&& !parent.as_os_str().is_empty()
|
||||
&& let Err(e) = std::fs::create_dir_all(parent)
|
||||
{
|
||||
log::error!("failed to create queue dir: {e}");
|
||||
}
|
||||
if let Ok(conn) = Connection::open(db_path)
|
||||
&& let Err(e) = ensure_schema(&conn)
|
||||
{
|
||||
log::error!("failed to initialize queue schema: {e}");
|
||||
}
|
||||
/// Wraps the shared DB pool; the schema is initialized once by
|
||||
/// [`crate::db::open_store`] (all three stores share the pool).
|
||||
pub fn new(pool: std::sync::Arc<crate::db::DbPool>) -> Self {
|
||||
Self {
|
||||
db_path: db_path.to_string(),
|
||||
pool,
|
||||
notify: Arc::new(Notify::new()),
|
||||
sweep_notify: Arc::new(Notify::new()),
|
||||
stop: Arc::new(AtomicBool::new(false)),
|
||||
worker: Mutex::new(Vec::new()),
|
||||
counter: AtomicU64::new(0),
|
||||
@@ -114,23 +114,50 @@ impl PersistentTaskQueue {
|
||||
let dead_letter: Arc<DeadLetter> =
|
||||
Arc::new(move |payload, message| Box::pin(dead_letter(payload, message)));
|
||||
self.recover_stale().await;
|
||||
let mut handles = Vec::with_capacity(QUEUE_WORKERS);
|
||||
let mut handles = Vec::with_capacity(QUEUE_WORKERS + 1);
|
||||
for _ in 0..QUEUE_WORKERS {
|
||||
let worker = QueueWorker {
|
||||
db_path: self.db_path.clone(),
|
||||
pool: std::sync::Arc::clone(&self.pool),
|
||||
notify: Arc::clone(&self.notify),
|
||||
stop: Arc::clone(&self.stop),
|
||||
handler: Arc::clone(&handler),
|
||||
dead_letter: Arc::clone(&dead_letter),
|
||||
};
|
||||
handles.push(tokio::spawn(worker.run_loop()));
|
||||
handles.push(tokio::spawn(worker.run_loop_supervised()));
|
||||
}
|
||||
// Periodic lease-expiry sweep: recovers rows a crashed/panicked
|
||||
// worker left `in_progress` (the lock TTL bounds the wait). Its own
|
||||
// notify (not the workers'): sharing that one let this task consume a
|
||||
// `notify_one` permit meant for a worker, which then slept through a
|
||||
// due row until some later event. Only `stop` wakes it.
|
||||
let sweep_pool = std::sync::Arc::clone(&self.pool);
|
||||
let sweep_notify = Arc::clone(&self.sweep_notify);
|
||||
let sweep_stop = Arc::clone(&self.stop);
|
||||
handles.push(tokio::spawn(async move {
|
||||
let mut interval = tokio::time::interval(Duration::from_secs(30));
|
||||
loop {
|
||||
let notified = sweep_notify.notified();
|
||||
tokio::pin!(notified);
|
||||
tokio::select! {
|
||||
_ = &mut notified => {}
|
||||
_ = interval.tick() => {}
|
||||
}
|
||||
if sweep_stop.load(Ordering::Relaxed) {
|
||||
break;
|
||||
}
|
||||
let result = sweep_pool.with_conn(move |conn| recover_update(conn)).await;
|
||||
if let Err(e) = result {
|
||||
log::error!("queue sweep failed: {e}");
|
||||
}
|
||||
}
|
||||
}));
|
||||
*self.worker.lock() = handles;
|
||||
}
|
||||
|
||||
pub async fn stop(&self) {
|
||||
self.stop.store(true, Ordering::Relaxed);
|
||||
self.notify.notify_waiters();
|
||||
self.sweep_notify.notify_waiters();
|
||||
let handles = std::mem::take(&mut *self.worker.lock());
|
||||
for handle in handles {
|
||||
let _ = handle.await;
|
||||
@@ -147,8 +174,8 @@ impl PersistentTaskQueue {
|
||||
self.counter.fetch_add(1, Ordering::Relaxed)
|
||||
);
|
||||
let payload = payload.to_string();
|
||||
log::info!("enqueued {id} (run_after {run_after:.1})");
|
||||
crate::db::with_conn(&self.db_path, move |conn| {
|
||||
log::debug!("enqueued {id} (run_after {run_after:.1})");
|
||||
self.pool.with_conn(move |conn| {
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO tasks (id, payload, run_after, attempts, status, locked_until, created_at) \
|
||||
VALUES (?1, ?2, ?3, 0, 'pending', 0, ?4)",
|
||||
@@ -157,21 +184,20 @@ impl PersistentTaskQueue {
|
||||
Ok(())
|
||||
})
|
||||
.await?;
|
||||
// Wake every sleeping worker: with several workers the one that finds
|
||||
// nothing due must not starve the newly inserted row.
|
||||
self.notify.notify_waiters();
|
||||
// `notify_one` stores a permit when no worker is registered, so a
|
||||
// notification fired between a worker's DB reads and its `notified()`
|
||||
// registration is not lost (notify_waiters would drop it). The
|
||||
// awakened worker re-leases and finds the new row.
|
||||
self.notify.notify_one();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn recover_stale(&self) {
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
conn.execute(
|
||||
"UPDATE tasks SET status='pending', locked_until=0 WHERE status='in_progress' AND locked_until < ?1",
|
||||
params![now_f64()],
|
||||
)?;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
self.recover_sweep().await;
|
||||
}
|
||||
|
||||
async fn recover_sweep(&self) {
|
||||
let result = self.pool.with_conn(move |conn| recover_update(conn)).await;
|
||||
if let Err(e) = result {
|
||||
log::error!("queue recovery failed: {e}");
|
||||
}
|
||||
@@ -179,11 +205,24 @@ impl PersistentTaskQueue {
|
||||
}
|
||||
|
||||
impl QueueWorker {
|
||||
/// Supervised worker: the inner loop runs in its own task so a panic
|
||||
/// (e.g. inside a handler or a DB closure) kills only that task; the
|
||||
/// supervisor respawns it until stop is set. The row a dead worker had
|
||||
/// leased is recovered by the periodic sweep once its lock TTL expires.
|
||||
async fn run_loop_supervised(self) {
|
||||
while !self.stop.load(Ordering::Relaxed) {
|
||||
let worker = self.clone();
|
||||
if let Err(e) = tokio::spawn(async move { worker.run_loop().await }).await {
|
||||
log::error!("queue worker panicked, restarting: {e}");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn run_loop(self) {
|
||||
while !self.stop.load(Ordering::Relaxed) {
|
||||
match self.lease_next().await {
|
||||
Some(row) => self.process(row).await,
|
||||
None => {
|
||||
Ok(Some(row)) => self.process(row).await,
|
||||
Ok(None) => {
|
||||
let wait_until = self.earliest_run_after().await;
|
||||
let notified = self.notify.notified();
|
||||
tokio::pin!(notified);
|
||||
@@ -200,13 +239,20 @@ impl QueueWorker {
|
||||
}
|
||||
}
|
||||
}
|
||||
// A lease failure while rows are due would otherwise loop
|
||||
// with sleep(0) and hammer SQLite; back off briefly.
|
||||
Err(e) => {
|
||||
log::error!("queue lease failed: {e}");
|
||||
tokio::time::sleep(Duration::from_secs(1)).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Leases the oldest due row (sets it `in_progress` with a lock TTL).
|
||||
async fn lease_next(&self) -> Option<LeasedRow> {
|
||||
let result = crate::db::with_conn(&self.db_path, |conn| {
|
||||
/// Errors are surfaced so the caller can back off instead of spinning.
|
||||
async fn lease_next(&self) -> Result<Option<LeasedRow>, rusqlite::Error> {
|
||||
self.pool.with_conn(|conn| {
|
||||
// BEGIN IMMEDIATE: with several workers, a deferred transaction
|
||||
// that read before another worker's lease commit would fail with
|
||||
// SQLITE_BUSY_SNAPSHOT. Taking the write lock up front serializes
|
||||
@@ -244,27 +290,22 @@ impl QueueWorker {
|
||||
attempts,
|
||||
}))
|
||||
})
|
||||
.await;
|
||||
match result {
|
||||
Ok(row) => row,
|
||||
Err(e) => {
|
||||
log::error!("queue lease failed: {e}");
|
||||
None
|
||||
}
|
||||
}
|
||||
.await
|
||||
}
|
||||
|
||||
async fn earliest_run_after(&self) -> Option<f64> {
|
||||
let result = crate::db::with_conn(&self.db_path, |conn| {
|
||||
let mut stmt =
|
||||
conn.prepare("SELECT MIN(run_after) FROM tasks WHERE status='pending'")?;
|
||||
let mut rows = stmt.query([])?;
|
||||
match rows.next()? {
|
||||
Some(row) => Ok(row.get::<_, Option<f64>>(0)?),
|
||||
None => Ok(None),
|
||||
}
|
||||
})
|
||||
.await;
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(|conn| {
|
||||
let mut stmt =
|
||||
conn.prepare("SELECT MIN(run_after) FROM tasks WHERE status='pending'")?;
|
||||
let mut rows = stmt.query([])?;
|
||||
match rows.next()? {
|
||||
Some(row) => Ok(row.get::<_, Option<f64>>(0)?),
|
||||
None => Ok(None),
|
||||
}
|
||||
})
|
||||
.await;
|
||||
match result {
|
||||
Ok(v) => v,
|
||||
Err(e) => {
|
||||
@@ -274,6 +315,11 @@ impl QueueWorker {
|
||||
}
|
||||
}
|
||||
|
||||
/// Processes one leased row, keeping the lease alive while the handler
|
||||
/// runs. Without the heartbeat a task longer than [`LOCK_TTL_SECONDS`]
|
||||
/// (slow download, ugoira encode, rate-limited batch forward) would have
|
||||
/// its lease expire mid-run; the expiry sweep would flip the row back to
|
||||
/// `pending` and another worker would process it again — duplicate sends.
|
||||
async fn process(&self, row: LeasedRow) {
|
||||
let payload: Value = match serde_json::from_str(&row.payload) {
|
||||
Ok(value) => value,
|
||||
@@ -284,10 +330,11 @@ impl QueueWorker {
|
||||
return;
|
||||
}
|
||||
};
|
||||
log::info!("processing {} (attempt {})", row.id, row.attempts + 1);
|
||||
match (self.handler)(payload).await {
|
||||
log::debug!("processing {} (attempt {})", row.id, row.attempts + 1);
|
||||
let outcome = self.run_with_lease(&row.id, payload).await;
|
||||
match outcome {
|
||||
Ok(()) => {
|
||||
log::info!("task {} completed", row.id);
|
||||
log::debug!("task {} completed", row.id);
|
||||
self.delete_row(&row.id).await;
|
||||
}
|
||||
Err(QueueError::Retryable {
|
||||
@@ -300,12 +347,13 @@ impl QueueWorker {
|
||||
self.delete_row(&row.id).await;
|
||||
(self.dead_letter)(payload, message).await;
|
||||
} else {
|
||||
log::info!(
|
||||
"task {} rescheduled in {delay_seconds:.1}s (attempt {})",
|
||||
let delay = scaled_retry_delay(delay_seconds, row.attempts);
|
||||
log::debug!(
|
||||
"task {} rescheduled in {delay:.1}s (attempt {})",
|
||||
row.id,
|
||||
row.attempts + 1
|
||||
);
|
||||
self.reschedule(&row.id, payload, delay_seconds, row.attempts + 1)
|
||||
self.reschedule(&row.id, payload, delay, row.attempts + 1)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
@@ -317,13 +365,51 @@ impl QueueWorker {
|
||||
}
|
||||
}
|
||||
|
||||
/// Drives the handler to completion, refreshing the row's `locked_until`
|
||||
/// every 30 s so the expiry sweep never re-leases a still-running task.
|
||||
/// The heartbeat is part of this future, not a separate spawned task: if
|
||||
/// the worker task dies (panic) the heartbeat dies with it and the sweep
|
||||
/// recovers the row exactly as before.
|
||||
async fn run_with_lease(&self, id: &str, payload: Value) -> Result<(), QueueError> {
|
||||
let fut = (self.handler)(payload);
|
||||
tokio::pin!(fut);
|
||||
let mut interval = tokio::time::interval(Duration::from_secs(30));
|
||||
// The first interval tick fires immediately; skip it (the lease was
|
||||
// just set by lease_next).
|
||||
interval.tick().await;
|
||||
let id_owned = id.to_string();
|
||||
loop {
|
||||
tokio::select! {
|
||||
result = &mut fut => return result,
|
||||
_ = interval.tick() => {
|
||||
let now = now_f64();
|
||||
let id = id_owned.clone();
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(move |conn| {
|
||||
conn.execute(
|
||||
"UPDATE tasks SET locked_until=?1 WHERE id=?2 AND status='in_progress'",
|
||||
params![now + LOCK_TTL_SECONDS, id],
|
||||
)
|
||||
})
|
||||
.await;
|
||||
if let Err(e) = result {
|
||||
log::error!("queue lease heartbeat failed: {e}");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn delete_row(&self, id: &str) {
|
||||
let id = id.to_string();
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
conn.execute("DELETE FROM tasks WHERE id = ?1", params![id])?;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(move |conn| {
|
||||
conn.execute("DELETE FROM tasks WHERE id = ?1", params![id])?;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
if let Err(e) = result {
|
||||
log::error!("queue delete failed: {e}");
|
||||
}
|
||||
@@ -332,7 +418,7 @@ impl QueueWorker {
|
||||
async fn reschedule(&self, id: &str, payload: Value, delay_seconds: f64, attempts: i32) {
|
||||
let id = id.to_string();
|
||||
let payload = payload.to_string();
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
let result = self.pool.with_conn(move |conn| {
|
||||
conn.execute(
|
||||
"UPDATE tasks SET payload=?1, run_after=?2, attempts=?3, status='pending', locked_until=0 WHERE id=?4",
|
||||
params![payload, now_f64() + delay_seconds, attempts, id],
|
||||
@@ -343,7 +429,8 @@ impl QueueWorker {
|
||||
if let Err(e) = result {
|
||||
log::error!("queue reschedule failed: {e}");
|
||||
}
|
||||
self.notify.notify_waiters();
|
||||
// Same permit semantics as enqueue: never lose the wakeup.
|
||||
self.notify.notify_one();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -352,10 +439,21 @@ mod tests {
|
||||
use super::*;
|
||||
use std::sync::atomic::{AtomicUsize, Ordering as AtomicOrdering};
|
||||
|
||||
#[test]
|
||||
fn scaled_retry_delay_scales_and_caps() {
|
||||
assert_eq!(scaled_retry_delay(1.0, 0), 1.0);
|
||||
assert_eq!(scaled_retry_delay(1.0, 1), 2.0);
|
||||
assert_eq!(scaled_retry_delay(1.0, 2), 4.0);
|
||||
assert_eq!(scaled_retry_delay(1.5, 1), 3.0);
|
||||
assert_eq!(scaled_retry_delay(1.0, 10), 300.0, "capped at 300s");
|
||||
assert_eq!(scaled_retry_delay(300.0, 0), 300.0);
|
||||
}
|
||||
|
||||
async fn new_queue() -> (PersistentTaskQueue, tempfile::TempDir) {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("queue.db");
|
||||
let queue = PersistentTaskQueue::new(path.to_str().unwrap());
|
||||
let pool = crate::db::open_store(path.to_str().unwrap()).unwrap();
|
||||
let queue = PersistentTaskQueue::new(pool);
|
||||
(queue, dir)
|
||||
}
|
||||
|
||||
@@ -462,10 +560,11 @@ mod tests {
|
||||
async fn stale_in_progress_row_is_recovered_on_start() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("queue.db");
|
||||
// Insert a stale leased row directly (lease expired).
|
||||
// Insert a stale leased row directly (lease expired). open_store runs
|
||||
// the schema; the queue below shares the same pool.
|
||||
let pool = crate::db::open_store(path.to_str().unwrap()).unwrap();
|
||||
{
|
||||
let conn = Connection::open(&path).unwrap();
|
||||
ensure_schema(&conn).unwrap();
|
||||
let conn = rusqlite::Connection::open(&path).unwrap();
|
||||
conn.execute(
|
||||
"INSERT INTO tasks (id, payload, run_after, attempts, status, locked_until, created_at) \
|
||||
VALUES ('task_stale', '{\"s\":1}', 0, 0, 'in_progress', ?1, 0)",
|
||||
@@ -473,7 +572,7 @@ mod tests {
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
let queue = PersistentTaskQueue::new(path.to_str().unwrap());
|
||||
let queue = PersistentTaskQueue::new(pool);
|
||||
let calls = Arc::new(AtomicUsize::new(0));
|
||||
let c = calls.clone();
|
||||
queue
|
||||
@@ -490,4 +589,40 @@ mod tests {
|
||||
assert_eq!(calls.load(AtomicOrdering::SeqCst), 1);
|
||||
queue.stop().await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn runtime_sweep_recovers_expired_lease() {
|
||||
let (queue, _dir) = new_queue().await;
|
||||
let calls = Arc::new(AtomicUsize::new(0));
|
||||
let c = calls.clone();
|
||||
queue
|
||||
.start(
|
||||
move |payload| {
|
||||
assert_eq!(payload["s"], 1);
|
||||
c.fetch_add(1, AtomicOrdering::SeqCst);
|
||||
async { Ok(()) }
|
||||
},
|
||||
|_payload, _message| async {},
|
||||
)
|
||||
.await;
|
||||
// Insert a stale leased row AFTER startup: without a runtime sweep it
|
||||
// would stay `in_progress` forever (only start() used to recover).
|
||||
{
|
||||
let conn = rusqlite::Connection::open(queue.pool.path()).unwrap();
|
||||
conn.execute(
|
||||
"INSERT INTO tasks (id, payload, run_after, attempts, status, locked_until, created_at) \
|
||||
VALUES ('task_stale_runtime', '{\"s\":1}', 0, 0, 'in_progress', ?1, 0)",
|
||||
params![now_f64() - 1000.0],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
queue.recover_sweep().await;
|
||||
tokio::time::sleep(Duration::from_millis(300)).await;
|
||||
assert_eq!(
|
||||
calls.load(AtomicOrdering::SeqCst),
|
||||
1,
|
||||
"expired lease must be recovered and processed exactly once"
|
||||
);
|
||||
queue.stop().await;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
//! Per-chat token-bucket rate limiting.
|
||||
//!
|
||||
//! Telegram throttles bots that burst past a chat's message budget
|
||||
//! (roughly 20 messages/min for channels/groups); today the bot absorbs
|
||||
//! those 429s with queue retries. This limiter smooths the burst *before*
|
||||
//! it reaches the API: media sends to a chat consume one token per
|
||||
//! message, refilled at [`REFILL_PER_SEC`], so a batch forward paces itself
|
||||
//! instead of tripping flood control. The queue retry stays as the safety
|
||||
//! net for limits this bucket does not model (global per-bot limits etc.).
|
||||
|
||||
use parking_lot::Mutex;
|
||||
use std::collections::HashMap;
|
||||
use std::sync::Arc;
|
||||
use std::sync::LazyLock;
|
||||
use std::time::Duration;
|
||||
|
||||
/// Burst capacity: how many messages may be sent at once without waiting.
|
||||
const CAPACITY: f64 = 20.0;
|
||||
/// Sustained refill: ~20 messages per minute.
|
||||
const REFILL_PER_SEC: f64 = 20.0 / 60.0;
|
||||
|
||||
struct State {
|
||||
/// Current token balance; may go negative (debt from an acquire larger
|
||||
/// than the capacity, repaid by subsequent refills).
|
||||
tokens: f64,
|
||||
last_refill: tokio::time::Instant,
|
||||
}
|
||||
|
||||
/// A token bucket: at most `CAPACITY` tokens accumulate, refilled at
|
||||
/// `REFILL_PER_SEC`. [`TokenBucket::acquire`] consumes `n` tokens, waiting
|
||||
/// for the deficit (a single acquire may exceed the capacity and goes into
|
||||
/// debt, which the refill repays).
|
||||
pub struct TokenBucket {
|
||||
capacity: f64,
|
||||
refill_per_sec: f64,
|
||||
state: Mutex<State>,
|
||||
}
|
||||
|
||||
impl TokenBucket {
|
||||
fn new(capacity: f64, refill_per_sec: f64) -> Self {
|
||||
TokenBucket {
|
||||
capacity,
|
||||
refill_per_sec,
|
||||
state: Mutex::new(State {
|
||||
tokens: capacity,
|
||||
last_refill: tokio::time::Instant::now(),
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
/// Applies the elapsed refill to `state`. Shared by [`Self::acquire`] and
|
||||
/// the idle check so the two cannot drift apart.
|
||||
fn refill(&self, state: &mut State) {
|
||||
let now = tokio::time::Instant::now();
|
||||
let elapsed = now
|
||||
.saturating_duration_since(state.last_refill)
|
||||
.as_secs_f64();
|
||||
// Refill up to the capacity; a debt (negative balance) is repaid
|
||||
// before any surplus accumulates.
|
||||
state.tokens = (state.tokens + elapsed * self.refill_per_sec).min(self.capacity);
|
||||
state.last_refill = now;
|
||||
}
|
||||
|
||||
/// Waits until `n` tokens are available, consuming them. The wait is
|
||||
/// bounded: the deficit is committed as debt and repaid over time, so a
|
||||
/// large acquire returns once its share of the refill budget has passed.
|
||||
pub async fn acquire(&self, n: f64) {
|
||||
// The parking_lot guard is confined to this block: only the plain
|
||||
// `wait` duration crosses the await (a guard across an await point
|
||||
// would make the future !Send).
|
||||
let wait = {
|
||||
let mut state = self.state.lock();
|
||||
self.refill(&mut state);
|
||||
if state.tokens >= n {
|
||||
state.tokens -= n;
|
||||
return;
|
||||
}
|
||||
// Commit the whole consumption now; the caller proceeds once the
|
||||
// deficit's worth of refill time has passed.
|
||||
let debt = n - state.tokens;
|
||||
state.tokens = -debt;
|
||||
debt / self.refill_per_sec
|
||||
};
|
||||
tokio::time::sleep(Duration::from_secs_f64(wait)).await;
|
||||
}
|
||||
|
||||
/// True when the bucket has refilled to capacity: no debt outstanding, so
|
||||
/// the chat has not sent anything recently.
|
||||
fn is_idle(&self) -> bool {
|
||||
let mut state = self.state.lock();
|
||||
self.refill(&mut state);
|
||||
state.tokens >= self.capacity
|
||||
}
|
||||
}
|
||||
|
||||
/// One limiter per chat, created on first use. Per-chat so one chat's burst
|
||||
/// never throttles another.
|
||||
static LIMITERS: LazyLock<Mutex<HashMap<i64, Arc<TokenBucket>>>> =
|
||||
LazyLock::new(|| Mutex::new(HashMap::new()));
|
||||
|
||||
/// Returns the shared limiter for a chat, creating it on first use.
|
||||
pub fn limiter_for(chat_id: i64) -> Arc<TokenBucket> {
|
||||
LIMITERS
|
||||
.lock()
|
||||
.entry(chat_id)
|
||||
.or_insert_with(|| Arc::new(TokenBucket::new(CAPACITY, REFILL_PER_SEC)))
|
||||
.clone()
|
||||
}
|
||||
|
||||
/// Drops limiters that are idle (refilled to capacity, so the chat has not
|
||||
/// sent recently) and are not still held by an in-flight sender. The map
|
||||
/// would otherwise keep one bucket per chat that ever sent media, forever.
|
||||
/// Called from the periodic sweep; returns how many were dropped.
|
||||
pub fn prune_idle() -> usize {
|
||||
let mut limiters = LIMITERS.lock();
|
||||
let before = limiters.len();
|
||||
// Lock order map → bucket, the only order taken anywhere.
|
||||
limiters.retain(|_, bucket| Arc::strong_count(bucket) > 1 || !bucket.is_idle());
|
||||
before - limiters.len()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn limiter_for_reuses_the_per_chat_bucket() {
|
||||
let a = limiter_for(1);
|
||||
let b = limiter_for(1);
|
||||
let c = limiter_for(2);
|
||||
assert!(Arc::ptr_eq(&a, &b), "same chat → same bucket");
|
||||
assert!(!Arc::ptr_eq(&a, &c), "different chat → different bucket");
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn burst_is_consumed_instantly_then_refill_waits() {
|
||||
let bucket = TokenBucket::new(3.0, 1.0);
|
||||
// A burst within capacity passes without waiting.
|
||||
bucket.acquire(3.0).await;
|
||||
// The bucket is empty now; one token needs 1s of refill.
|
||||
let start = tokio::time::Instant::now();
|
||||
bucket.acquire(1.0).await;
|
||||
assert!(
|
||||
start.elapsed() >= Duration::from_secs(1),
|
||||
"elapsed {:?}",
|
||||
start.elapsed()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn acquire_larger_than_capacity_waits_for_the_deficit() {
|
||||
let bucket = TokenBucket::new(2.0, 1.0);
|
||||
// 5 tokens with a capacity of 2: the 3-token deficit takes 3s.
|
||||
let start = tokio::time::Instant::now();
|
||||
bucket.acquire(5.0).await;
|
||||
assert!(
|
||||
start.elapsed() >= Duration::from_secs(3),
|
||||
"elapsed {:?}",
|
||||
start.elapsed()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn prune_idle_drops_full_unheld_buckets_only() {
|
||||
// Held by this task: kept even at full capacity, a sender has it.
|
||||
let held = limiter_for(9_001);
|
||||
assert!(held.is_idle(), "a fresh bucket is full");
|
||||
// Only the map holds this one and it is full → dropped.
|
||||
limiter_for(9_002);
|
||||
// Mid-debt (an acquire larger than the capacity): kept.
|
||||
{
|
||||
let bucket = Arc::new(TokenBucket::new(CAPACITY, REFILL_PER_SEC));
|
||||
bucket.state.lock().tokens = -1.0;
|
||||
LIMITERS.lock().insert(9_003, bucket);
|
||||
}
|
||||
|
||||
assert!(prune_idle() >= 1);
|
||||
|
||||
let limiters = LIMITERS.lock();
|
||||
assert!(limiters.contains_key(&9_001), "held bucket pruned");
|
||||
assert!(!limiters.contains_key(&9_002), "idle unheld bucket kept");
|
||||
assert!(limiters.contains_key(&9_003), "indebted bucket pruned");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,128 @@
|
||||
//! Payload → Telegram input types: `InputFile` selection (cached file id /
|
||||
//! URL / local path), the per-kind `InputMedia` builders and the media-group
|
||||
//! assembly with its caption rule.
|
||||
|
||||
use super::MediaItemPayload;
|
||||
use teloxide::types::{
|
||||
InputFile, InputMedia, InputMediaAnimation, InputMediaPhoto, InputMediaVideo, ParseMode,
|
||||
};
|
||||
|
||||
fn parse_media_url(s: &str) -> Result<url::Url, String> {
|
||||
url::Url::parse(s).map_err(|e| format!("invalid media URL: {e}"))
|
||||
}
|
||||
|
||||
pub(super) fn item_url(item: &MediaItemPayload) -> &str {
|
||||
match item {
|
||||
MediaItemPayload::Photo { media, .. }
|
||||
| MediaItemPayload::Video { media, .. }
|
||||
| MediaItemPayload::Animation { media, .. } => media,
|
||||
}
|
||||
}
|
||||
|
||||
/// Remote http(s) URLs are handed to Telegram to fetch; everything else
|
||||
/// (e.g. a locally encoded ugoira MP4) is uploaded directly.
|
||||
pub(super) fn input_file_for(media: &str) -> Result<InputFile, String> {
|
||||
if media.starts_with("http://") || media.starts_with("https://") {
|
||||
Ok(InputFile::url(parse_media_url(media)?))
|
||||
} else if !std::path::Path::new(media).exists() {
|
||||
// A retried task may reference a temp file the original send's
|
||||
// TempDir already cleaned up; fail fast and permanent instead of
|
||||
// burning retries on a file that can never come back.
|
||||
Err(format!("local media file missing: {media}"))
|
||||
} else {
|
||||
Ok(InputFile::file(media))
|
||||
}
|
||||
}
|
||||
|
||||
impl MediaItemPayload {
|
||||
/// The input for a send: a cached file id goes out as `InputFile::file_id`
|
||||
/// (no fetch, no upload), URLs go to Telegram, anything else is a local
|
||||
/// path (transient upload fallback).
|
||||
fn input_file(&self) -> Result<InputFile, String> {
|
||||
match self {
|
||||
MediaItemPayload::Photo {
|
||||
media,
|
||||
file_id: true,
|
||||
..
|
||||
}
|
||||
| MediaItemPayload::Video {
|
||||
media,
|
||||
file_id: true,
|
||||
..
|
||||
}
|
||||
| MediaItemPayload::Animation {
|
||||
media,
|
||||
file_id: true,
|
||||
..
|
||||
} => Ok(InputFile::file_id(media.clone().into())),
|
||||
_ => input_file_for(item_url(self)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn photo_media(file: InputFile, caption: Option<&str>, spoiler: bool) -> InputMedia {
|
||||
let mut photo = InputMediaPhoto::new(file).parse_mode(ParseMode::Html);
|
||||
if let Some(caption) = caption {
|
||||
photo = photo.caption(caption);
|
||||
}
|
||||
if spoiler {
|
||||
photo = photo.spoiler();
|
||||
}
|
||||
InputMedia::Photo(photo)
|
||||
}
|
||||
|
||||
pub(super) fn video_media(file: InputFile, caption: Option<&str>, spoiler: bool) -> InputMedia {
|
||||
let mut video = InputMediaVideo::new(file).parse_mode(ParseMode::Html);
|
||||
if let Some(caption) = caption {
|
||||
video = video.caption(caption);
|
||||
}
|
||||
if spoiler {
|
||||
video = video.spoiler();
|
||||
}
|
||||
InputMedia::Video(video)
|
||||
}
|
||||
|
||||
pub(super) fn animation_media(file: InputFile, caption: Option<&str>, spoiler: bool) -> InputMedia {
|
||||
let mut animation = InputMediaAnimation::new(file).parse_mode(ParseMode::Html);
|
||||
if let Some(caption) = caption {
|
||||
animation = animation.caption(caption);
|
||||
}
|
||||
if spoiler {
|
||||
animation = animation.spoiler();
|
||||
}
|
||||
InputMedia::Animation(animation)
|
||||
}
|
||||
|
||||
/// Builds a media group from payloads; only the first item of the batch gets
|
||||
/// the caption (Telegram rejects captions on later items).
|
||||
pub(super) fn build_media_group(
|
||||
batch: &[MediaItemPayload],
|
||||
caption: Option<&str>,
|
||||
) -> Result<Vec<InputMedia>, String> {
|
||||
batch
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, item)| {
|
||||
let item_caption = if i == 0 { caption } else { None };
|
||||
Ok(match item {
|
||||
MediaItemPayload::Photo { has_spoiler, .. } => {
|
||||
photo_media(item.input_file()?, item_caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Video {
|
||||
has_spoiler,
|
||||
thumbnail,
|
||||
..
|
||||
} => {
|
||||
let mut video = video_media(item.input_file()?, item_caption, *has_spoiler);
|
||||
if let (Some(thumb), InputMedia::Video(v)) = (thumbnail, &mut video) {
|
||||
*v = v.clone().thumbnail(input_file_for(thumb)?);
|
||||
}
|
||||
video
|
||||
}
|
||||
MediaItemPayload::Animation { has_spoiler, .. } => {
|
||||
animation_media(item.input_file()?, item_caption, *has_spoiler)
|
||||
}
|
||||
})
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,366 @@
|
||||
//! Everything around a send: the link-cache write that follows one, the
|
||||
//! keep-alive registry for locally produced media, task settlement, the
|
||||
//! post-send actions (edit prompt / channel forward) and the queue entry
|
||||
//! points.
|
||||
|
||||
use super::{SendError, Task, forward_messages, send_animation, send_media_sequence};
|
||||
use crate::ctx::AppContext;
|
||||
use crate::db::{now_f64, unix_now};
|
||||
use crate::handlers::log_key;
|
||||
use crate::link_cache::{CachedMedia, CachedMediaKind, LinkCache};
|
||||
use crate::media_sender::MediaSender;
|
||||
use crate::queue::{PersistentTaskQueue, QueueError};
|
||||
use crate::state::EditMessage;
|
||||
use std::collections::HashMap;
|
||||
use std::sync::LazyLock;
|
||||
use teloxide::types::{ChatId, InlineKeyboardButton, InlineKeyboardMarkup, Message, MessageId};
|
||||
|
||||
/// Persists a successful send under the post's cache key. Only runs for a
|
||||
/// fresh (non-resumed) task that carried raw cache data with no file ids yet.
|
||||
pub(super) async fn cache_sent_task(ctx: &AppContext<'_>, task: &Task, media: Vec<CachedMedia>) {
|
||||
let Some(cache_data) = task.cache_data() else {
|
||||
return;
|
||||
};
|
||||
if !cache_data.media.is_empty() || media.is_empty() {
|
||||
return;
|
||||
}
|
||||
let mut post = cache_data.clone();
|
||||
post.media = media;
|
||||
if let Some(key) = x_media::site::cache_key(&post.url) {
|
||||
ctx.link_cache.put(&key, &post).await;
|
||||
log::debug!("cached send for [key={}]", log_key(&post.url));
|
||||
}
|
||||
}
|
||||
|
||||
/// Persists a lone animation send under the post's cache key.
|
||||
pub(super) async fn cache_animation_send(ctx: &AppContext<'_>, task: &Task, message: &Message) {
|
||||
if let Some(file_id) = message.animation().map(|a| a.file.id.to_string()) {
|
||||
cache_sent_task(
|
||||
ctx,
|
||||
task,
|
||||
vec![CachedMedia {
|
||||
kind: CachedMediaKind::Animation,
|
||||
file_id,
|
||||
}],
|
||||
)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
|
||||
/// How a task ended. The two states differ only in whether a link-cache entry
|
||||
/// may still be holding the (now unusable) media.
|
||||
pub(crate) enum Settled {
|
||||
Sent,
|
||||
Failed,
|
||||
}
|
||||
|
||||
/// Every path that ends a task's life — sent, permanently failed, or
|
||||
/// dead-lettered after the last retry — funnels through here, so the cleanup a
|
||||
/// settled task owes cannot be forgotten by a new path: release the keep-alive
|
||||
/// temp media (retryable tasks keep it, they will be resent) and drop the
|
||||
/// link-cache entry that a failed send's stale file ids would keep poisoning.
|
||||
pub(crate) async fn settle_task(ctx: &AppContext<'_>, task: &Task, outcome: Settled) {
|
||||
if matches!(outcome, Settled::Failed) {
|
||||
invalidate_cache(ctx.link_cache, task).await;
|
||||
}
|
||||
release_keep_alive(task);
|
||||
}
|
||||
|
||||
/// A cached Telegram file id failed permanently (stale/expired); drop the
|
||||
/// cache entry so the next request re-fetches instead of repeating it.
|
||||
async fn invalidate_cache(cache: &LinkCache, task: &Task) {
|
||||
if task.is_cached_send()
|
||||
&& let Some(url) = task.source_url()
|
||||
&& let Some(key) = x_media::site::cache_key(url)
|
||||
{
|
||||
log::debug!("removing stale link cache entry for [key={}]", log_key(url));
|
||||
cache.remove(&key).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// Locally produced media files (ugoira MP4, bsky remux MP4) whose temp dirs
|
||||
/// must stay alive while their task may be retried by the queue. The fetch
|
||||
/// pipeline hands ownership here via
|
||||
/// [`x_media::site::Fetched::take_keep_alive`] before that
|
||||
/// [`x_media::site::Fetched`] is dropped; a queued retry runs after that drop,
|
||||
/// so without this the local file would be gone by the time the retry sends
|
||||
/// it. Entries are removed when the task settles (see [`release_keep_alive`]).
|
||||
pub(crate) static KEEP_ALIVE: LazyLock<parking_lot::Mutex<Vec<tempfile::TempDir>>> =
|
||||
LazyLock::new(|| parking_lot::Mutex::new(Vec::new()));
|
||||
|
||||
/// Drops the keep-alive temp dirs holding media referenced by `task` (matched
|
||||
/// by path prefix). Called once a task settles — sent or permanently failed —
|
||||
/// so retry-only temp files do not leak; retryable tasks keep them alive.
|
||||
pub(crate) fn release_keep_alive(task: &Task) {
|
||||
let paths = task.local_media_paths();
|
||||
if paths.is_empty() {
|
||||
return;
|
||||
}
|
||||
let mut alive = KEEP_ALIVE.lock();
|
||||
alive.retain(|dir| {
|
||||
let dir_path = dir.path();
|
||||
!paths.iter().any(|p| p.starts_with(dir_path))
|
||||
});
|
||||
}
|
||||
|
||||
/// One button per template name (column layout), then the confirm button.
|
||||
/// Sorted by name: the templates live in a `HashMap`, so an unsorted walk
|
||||
/// would reshuffle the buttons between prompts.
|
||||
pub(super) fn build_edit_markup(templates: &HashMap<String, String>) -> InlineKeyboardMarkup {
|
||||
let mut names: Vec<&String> = templates.keys().collect();
|
||||
names.sort();
|
||||
let mut rows = Vec::with_capacity(names.len() + 1);
|
||||
for name in names {
|
||||
rows.push(vec![InlineKeyboardButton::callback(
|
||||
name.clone(),
|
||||
format!("template|{name}"),
|
||||
)]);
|
||||
}
|
||||
rows.push(vec![InlineKeyboardButton::callback(
|
||||
"↩️ Confirm",
|
||||
"forward",
|
||||
)]);
|
||||
InlineKeyboardMarkup::new(rows)
|
||||
}
|
||||
|
||||
/// Notifies a chat about a dead-lettered task (skips when `notify_chat_id` is
|
||||
/// absent).
|
||||
pub(super) async fn notify_failure(
|
||||
sender: &dyn MediaSender,
|
||||
chat_id: Option<i64>,
|
||||
message_id: Option<i64>,
|
||||
message: &str,
|
||||
) {
|
||||
let Some(chat_id) = chat_id else { return };
|
||||
let reply_to = message_id.map(|id| MessageId(id as i32));
|
||||
if let Err(e) = sender
|
||||
.send_message(ChatId(chat_id), message.to_string(), reply_to, None)
|
||||
.await
|
||||
{
|
||||
log::error!("failed to notify about failed task: {e}");
|
||||
}
|
||||
}
|
||||
|
||||
/// After a successful send: either open the edit-before-forward prompt or
|
||||
/// forward to the configured channel (with retry/queue handling).
|
||||
pub(crate) async fn post_send_actions(ctx: &AppContext<'_>, task: &Task, message_ids: Vec<i64>) {
|
||||
let (
|
||||
chat_id,
|
||||
reply_to,
|
||||
source_url,
|
||||
edit_before_forward,
|
||||
forward_channel_id,
|
||||
notify_chat_id,
|
||||
notify_message_id,
|
||||
) = match task {
|
||||
Task::SendMediaSequence {
|
||||
chat_id,
|
||||
reply_to_message_id,
|
||||
source_url,
|
||||
edit_before_forward,
|
||||
forward_channel_id,
|
||||
notify_chat_id,
|
||||
notify_message_id,
|
||||
..
|
||||
}
|
||||
| Task::SendAnimation {
|
||||
chat_id,
|
||||
reply_to_message_id,
|
||||
source_url,
|
||||
edit_before_forward,
|
||||
forward_channel_id,
|
||||
notify_chat_id,
|
||||
notify_message_id,
|
||||
..
|
||||
} => (
|
||||
*chat_id,
|
||||
*reply_to_message_id,
|
||||
source_url.clone(),
|
||||
*edit_before_forward,
|
||||
*forward_channel_id,
|
||||
*notify_chat_id,
|
||||
*notify_message_id,
|
||||
),
|
||||
Task::ForwardMessages { .. } => return,
|
||||
};
|
||||
|
||||
if edit_before_forward {
|
||||
let keyboard = build_edit_markup(&ctx.chat_store.get(chat_id).await.template);
|
||||
let prompt = ctx
|
||||
.sender
|
||||
.send_message(
|
||||
ChatId(chat_id),
|
||||
"Reply to edit message.".to_string(),
|
||||
Some(MessageId(reply_to as i32)),
|
||||
Some(keyboard),
|
||||
)
|
||||
.await;
|
||||
match prompt {
|
||||
Ok(prompt_id) => {
|
||||
log::info!(
|
||||
"edit-before-forward prompt {prompt_id} opened for {} message(s)",
|
||||
message_ids.len()
|
||||
);
|
||||
let source_url = source_url.clone();
|
||||
ctx.chat_store
|
||||
.update(chat_id, move |data| {
|
||||
data.edit_message.insert(
|
||||
prompt_id,
|
||||
EditMessage {
|
||||
url: source_url,
|
||||
chat_id,
|
||||
forward_message_ids: message_ids,
|
||||
template: String::new(),
|
||||
created_at: unix_now(),
|
||||
},
|
||||
);
|
||||
})
|
||||
.await;
|
||||
}
|
||||
Err(e) => log::error!("failed to send edit prompt: {e}"),
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if let Some(channel_id) = forward_channel_id {
|
||||
log::info!(
|
||||
"forwarding {} message(s) to channel {channel_id}",
|
||||
message_ids.len()
|
||||
);
|
||||
let forward_task = Task::ForwardMessages {
|
||||
from_chat_id: chat_id,
|
||||
to_chat_id: channel_id,
|
||||
message_ids,
|
||||
notify_chat_id,
|
||||
notify_message_id,
|
||||
};
|
||||
match forward_messages(ctx, &forward_task).await {
|
||||
Ok(()) => {}
|
||||
Err(SendError::Retryable {
|
||||
delay_seconds,
|
||||
task,
|
||||
}) => {
|
||||
enqueue_retry(ctx.task_queue, *task, delay_seconds).await;
|
||||
}
|
||||
Err(SendError::Permanent { message, .. }) => {
|
||||
notify_failure(
|
||||
ctx.sender,
|
||||
notify_chat_id,
|
||||
notify_message_id,
|
||||
&format!("Task failed after retries: {message}"),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Enqueues a task for a later attempt (retry / forward resume). When the
|
||||
/// enqueue itself fails the task can never be sent again, so its keep-alive
|
||||
/// temp media is released instead of leaking until process exit.
|
||||
pub(crate) async fn enqueue_retry(queue: &PersistentTaskQueue, task: Task, delay_seconds: f64) {
|
||||
let payload = serde_json::to_value(&task).expect("task serializes");
|
||||
let run_after = now_f64() + delay_seconds;
|
||||
if let Err(e) = queue.enqueue(payload, run_after).await {
|
||||
log::error!("failed to enqueue retry: {e}");
|
||||
release_keep_alive(&task);
|
||||
}
|
||||
}
|
||||
|
||||
/// Queue entry point: parses the stored task and dispatches.
|
||||
pub(crate) async fn handle_task(
|
||||
ctx: &AppContext<'_>,
|
||||
payload: serde_json::Value,
|
||||
) -> Result<(), QueueError> {
|
||||
let task: Task = match serde_json::from_value(payload.clone()) {
|
||||
Ok(task) => task,
|
||||
Err(e) => {
|
||||
return Err(QueueError::Permanent {
|
||||
message: format!("invalid task payload: {e}"),
|
||||
payload,
|
||||
});
|
||||
}
|
||||
};
|
||||
match task {
|
||||
Task::SendMediaSequence { .. } | Task::SendAnimation { .. } => {
|
||||
let message_ids = match send_media_or_animation(ctx, &task).await {
|
||||
Ok(ids) => ids,
|
||||
Err(SendError::Retryable {
|
||||
delay_seconds,
|
||||
task,
|
||||
}) => {
|
||||
return Err(QueueError::Retryable {
|
||||
delay_seconds,
|
||||
payload: serde_json::to_value(task).expect("task serializes"),
|
||||
});
|
||||
}
|
||||
Err(SendError::Permanent { message, task }) => {
|
||||
settle_task(ctx, &task, Settled::Failed).await;
|
||||
return Err(QueueError::Permanent {
|
||||
message,
|
||||
payload: serde_json::to_value(task).expect("task serializes"),
|
||||
});
|
||||
}
|
||||
};
|
||||
// A task only reaches the queue after a failed send, so this
|
||||
// successful run is the first time post_send_actions can fire —
|
||||
// the fresh attempt failed before it ever got here. Run it
|
||||
// unconditionally: `post_send_actions` executes once, after the
|
||||
// whole sequence (every batch) completed, so the channel forward
|
||||
// and the edit-before-forward prompt must not be lost just
|
||||
// because the send needed a retry.
|
||||
post_send_actions(ctx, &task, message_ids).await;
|
||||
settle_task(ctx, &task, Settled::Sent).await;
|
||||
Ok(())
|
||||
}
|
||||
Task::ForwardMessages { .. } => match forward_messages(ctx, &task).await {
|
||||
Ok(()) => Ok(()),
|
||||
Err(SendError::Retryable {
|
||||
delay_seconds,
|
||||
task,
|
||||
}) => Err(QueueError::Retryable {
|
||||
delay_seconds,
|
||||
payload: serde_json::to_value(task).expect("task serializes"),
|
||||
}),
|
||||
Err(SendError::Permanent { message, task }) => {
|
||||
settle_task(ctx, &task, Settled::Failed).await;
|
||||
Err(QueueError::Permanent {
|
||||
message,
|
||||
payload: serde_json::to_value(task).expect("task serializes"),
|
||||
})
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
async fn send_media_or_animation(ctx: &AppContext<'_>, task: &Task) -> Result<Vec<i64>, SendError> {
|
||||
match task {
|
||||
Task::SendMediaSequence { .. } => send_media_sequence(ctx, task).await,
|
||||
Task::SendAnimation { .. } => send_animation(ctx, task).await,
|
||||
Task::ForwardMessages { .. } => unreachable!(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Dead-letter callback wired to the queue in main: settles the task and
|
||||
/// notifies its chat.
|
||||
pub(crate) async fn dead_letter_notify(
|
||||
ctx: &AppContext<'_>,
|
||||
payload: serde_json::Value,
|
||||
message: String,
|
||||
) {
|
||||
// A dead-lettered task never runs again, and the queue dead-letters retry
|
||||
// exhaustion itself (the handler is not called again), so this is the only
|
||||
// place that sees the final payload.
|
||||
if let Ok(task) = serde_json::from_value::<Task>(payload.clone()) {
|
||||
settle_task(ctx, &task, Settled::Failed).await;
|
||||
}
|
||||
let notify_chat_id = payload.get("notify_chat_id").and_then(|v| v.as_i64());
|
||||
let notify_message_id = payload.get("notify_message_id").and_then(|v| v.as_i64());
|
||||
notify_failure(
|
||||
ctx.sender,
|
||||
notify_chat_id,
|
||||
notify_message_id,
|
||||
&format!("Task failed after retries: {message}"),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
@@ -0,0 +1,345 @@
|
||||
//! Download-and-reupload fallback: when Telegram cannot fetch a media URL
|
||||
//! itself (hotlink protection), the bot downloads the file, shrinks photos
|
||||
//! that exceed Telegram's limits and uploads the batch via multipart.
|
||||
|
||||
use super::input_media::{animation_media, input_file_for, item_url, photo_media, video_media};
|
||||
use super::{MediaItemPayload, SendError, Task, classify_to_send_error, retry_delay_seconds};
|
||||
use crate::media_sender::MediaSender;
|
||||
use crate::photo::{self, MAX_UPLOAD_BYTES, PhotoPrep};
|
||||
use teloxide::prelude::*;
|
||||
use teloxide::types::{ChatId, InputFile, InputMedia, MessageId};
|
||||
use tempfile::NamedTempFile;
|
||||
use x_media::site::FetchError;
|
||||
|
||||
/// Infers a file extension from magic bytes so Telegram detects the mime type
|
||||
/// on multipart uploads.
|
||||
pub(super) fn sniff_ext(bytes: &[u8]) -> &'static str {
|
||||
if bytes.starts_with(&[0xFF, 0xD8]) {
|
||||
"jpg"
|
||||
} else if bytes.starts_with(b"\x89PNG") {
|
||||
"png"
|
||||
} else if bytes.starts_with(b"RIFF") && bytes.len() >= 12 && &bytes[8..12] == b"WEBP" {
|
||||
"webp"
|
||||
} else if bytes.starts_with(b"GIF8") {
|
||||
"gif"
|
||||
} else if bytes.len() >= 12 && &bytes[4..8] == b"ftyp" {
|
||||
"mp4"
|
||||
} else {
|
||||
"bin"
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) enum FallbackError {
|
||||
Retryable {
|
||||
delay_seconds: f64,
|
||||
},
|
||||
Permanent {
|
||||
message: String,
|
||||
},
|
||||
/// The downloaded file exceeds the upload cap; the caller falls back to
|
||||
/// the item's smaller URL.
|
||||
MediaTooLarge,
|
||||
}
|
||||
|
||||
/// Brings a downloaded photo within Telegram's limits via the pure-Rust
|
||||
/// chain in [`crate::photo`] (no ffmpeg): dimension cap / upload cap
|
||||
/// exceeded photos are decoded, downscaled with Lanczos3, PNG bit depth
|
||||
/// reduced (>24-bit → 24-bit RGB, ≤24-bit untouched) and transcoded to JPEG
|
||||
/// only if still too big. Anything that cannot be fixed falls back to the
|
||||
/// item's smaller URL.
|
||||
///
|
||||
/// Downloads one media item to a temp file (deleted on drop), returning the
|
||||
/// file plus the downloaded bytes (photos keep the bytes for
|
||||
/// [`photo::prepare_photo`] — re-reading the file would double the I/O).
|
||||
/// Network errors are retryable; size over the upload cap and other download
|
||||
/// errors are not.
|
||||
async fn download_to_temp(
|
||||
item: &MediaItemPayload,
|
||||
) -> Result<(NamedTempFile, bytes::Bytes), FallbackError> {
|
||||
let media_url = match item {
|
||||
MediaItemPayload::Photo { media, .. }
|
||||
| MediaItemPayload::Video { media, .. }
|
||||
| MediaItemPayload::Animation { media, .. } => media,
|
||||
};
|
||||
// Photos are downloaded even over the upload cap so `prepare_photo` can
|
||||
// downscale / transcode them (cap = decode budget); videos/animations
|
||||
// abort as soon as the upload cap is crossed mid-stream.
|
||||
let limit = if matches!(item, MediaItemPayload::Photo { .. }) {
|
||||
photo::MAX_DECODE_BYTES
|
||||
} else {
|
||||
MAX_UPLOAD_BYTES + 1
|
||||
};
|
||||
let bytes = match x_media::site::download_media_limited(media_url, limit).await {
|
||||
Ok(bytes) => bytes,
|
||||
Err(FetchError::Http(_)) => {
|
||||
return Err(FallbackError::Retryable {
|
||||
delay_seconds: retry_delay_seconds(0),
|
||||
});
|
||||
}
|
||||
Err(FetchError::TooLarge) => {
|
||||
return Err(FallbackError::MediaTooLarge);
|
||||
}
|
||||
Err(e) => {
|
||||
return Err(FallbackError::Permanent {
|
||||
message: format!("download failed: {e}"),
|
||||
});
|
||||
}
|
||||
};
|
||||
let ext = sniff_ext(&bytes);
|
||||
let mut file = tempfile::Builder::new()
|
||||
.suffix(&format!(".{ext}"))
|
||||
.tempfile()
|
||||
.map_err(|e| FallbackError::Permanent {
|
||||
message: format!("temp file failed: {e}"),
|
||||
})?;
|
||||
use std::io::Write;
|
||||
file.as_file_mut()
|
||||
.write_all(&bytes)
|
||||
.map_err(|e| FallbackError::Permanent {
|
||||
message: format!("temp file write failed: {e}"),
|
||||
})?;
|
||||
Ok((file, bytes))
|
||||
}
|
||||
|
||||
/// Builds the media group item from an uploaded file.
|
||||
fn media_from_file(
|
||||
item: &MediaItemPayload,
|
||||
path: std::path::PathBuf,
|
||||
caption: Option<&str>,
|
||||
thumbnail: Option<&str>,
|
||||
) -> Result<InputMedia, String> {
|
||||
let mut media = match item {
|
||||
MediaItemPayload::Photo { has_spoiler, .. } => {
|
||||
photo_media(InputFile::file(path), caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Video { has_spoiler, .. } => {
|
||||
video_media(InputFile::file(path), caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Animation { has_spoiler, .. } => {
|
||||
animation_media(InputFile::file(path), caption, *has_spoiler)
|
||||
}
|
||||
};
|
||||
if let (Some(thumb), InputMedia::Video(v)) = (thumbnail, &mut media) {
|
||||
*v = v.clone().thumbnail(input_file_for(thumb)?);
|
||||
}
|
||||
Ok(media)
|
||||
}
|
||||
|
||||
/// Builds the media group item from a (smaller) URL.
|
||||
fn media_from_url(
|
||||
item: &MediaItemPayload,
|
||||
url: &str,
|
||||
caption: Option<&str>,
|
||||
thumbnail: Option<&str>,
|
||||
) -> Result<InputMedia, String> {
|
||||
let mut media = match item {
|
||||
MediaItemPayload::Photo { has_spoiler, .. } => {
|
||||
photo_media(input_file_for(url)?, caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Video { has_spoiler, .. } => {
|
||||
video_media(input_file_for(url)?, caption, *has_spoiler)
|
||||
}
|
||||
MediaItemPayload::Animation { has_spoiler, .. } => {
|
||||
animation_media(input_file_for(url)?, caption, *has_spoiler)
|
||||
}
|
||||
};
|
||||
if let (Some(thumb), InputMedia::Video(v)) = (thumbnail, &mut media) {
|
||||
*v = v.clone().thumbnail(input_file_for(thumb)?);
|
||||
}
|
||||
Ok(media)
|
||||
}
|
||||
|
||||
/// One item prepared for the upload fallback: the ready-to-send media plus
|
||||
/// the temp file that must stay on disk until the group request completes.
|
||||
pub(super) struct PreparedItem {
|
||||
/// Original position in the batch (concurrent prep completes out of order).
|
||||
pub(super) index: usize,
|
||||
pub(super) media: InputMedia,
|
||||
pub(super) keep_alive: Option<NamedTempFile>,
|
||||
}
|
||||
|
||||
/// Downloads / processes one media item for the upload fallback (see
|
||||
/// [`send_batch_via_upload`]). Local files are uploaded directly; oversized
|
||||
/// items fall back to their smaller URL; photos are downscaled/transcoded.
|
||||
pub(super) async fn prepare_upload_item(
|
||||
item: MediaItemPayload,
|
||||
index: usize,
|
||||
caption: Option<&str>,
|
||||
) -> Result<PreparedItem, FallbackError> {
|
||||
// Locally produced files (ugoira / bsky remux MP4): nothing to download
|
||||
// or shrink — upload the file directly. The send is a multipart upload,
|
||||
// so the only remaining failure is an upload-cap error, which is
|
||||
// permanent (a video cannot be re-encoded here).
|
||||
let media_url = item_url(&item);
|
||||
if !media_url.starts_with("http://") && !media_url.starts_with("https://") {
|
||||
let media = media_from_file(
|
||||
&item,
|
||||
std::path::PathBuf::from(media_url),
|
||||
caption,
|
||||
item.thumbnail_url(),
|
||||
)
|
||||
.map_err(|message| FallbackError::Permanent { message })?;
|
||||
return Ok(PreparedItem {
|
||||
index,
|
||||
media,
|
||||
keep_alive: None,
|
||||
});
|
||||
}
|
||||
// Size check before downloading/uploading: over the cap, use the
|
||||
// smaller URL instead of the file. Photos are exempt — they are
|
||||
// downloaded and processed (downscale / PNG→JPEG) before uploading.
|
||||
let too_large = match x_media::site::media_size(media_url).await {
|
||||
Ok(Some(size)) => size > MAX_UPLOAD_BYTES,
|
||||
_ => false,
|
||||
};
|
||||
let too_large = too_large && !matches!(item, MediaItemPayload::Photo { .. });
|
||||
if too_large {
|
||||
let url = item
|
||||
.fallback_url()
|
||||
.ok_or_else(|| FallbackError::Permanent {
|
||||
message: "media too large".into(),
|
||||
})?;
|
||||
let media = media_from_url(&item, url, caption, item.thumbnail_url())
|
||||
.map_err(|message| FallbackError::Permanent { message })?;
|
||||
return Ok(PreparedItem {
|
||||
index,
|
||||
media,
|
||||
keep_alive: None,
|
||||
});
|
||||
}
|
||||
match download_to_temp(&item).await {
|
||||
Ok((file, bytes)) => {
|
||||
if matches!(item, MediaItemPayload::Photo { .. }) {
|
||||
// Telegram rejects photos wider+taller than 10000 px combined
|
||||
// (PHOTO_INVALID_DIMENSIONS): downscale the downloaded file
|
||||
// before uploading; photos that cannot be brought within the
|
||||
// limits degrade to the smaller URL. CPU-heavy work runs off
|
||||
// the async executor thread.
|
||||
let prep = tokio::task::spawn_blocking(move || photo::prepare_photo(file, &bytes))
|
||||
.await
|
||||
.map_err(|e| FallbackError::Permanent {
|
||||
message: format!("photo worker panicked: {e}"),
|
||||
})?
|
||||
.map_err(|message| FallbackError::Permanent { message })?;
|
||||
match prep {
|
||||
PhotoPrep::Upload(upload) => {
|
||||
let path = upload.path().to_path_buf();
|
||||
let media = media_from_file(&item, path, caption, item.thumbnail_url())
|
||||
.map_err(|message| FallbackError::Permanent { message })?;
|
||||
Ok(PreparedItem {
|
||||
index,
|
||||
media,
|
||||
keep_alive: Some(upload),
|
||||
})
|
||||
}
|
||||
PhotoPrep::UseFallback => {
|
||||
let url = item.fallback_url().ok_or_else(|| FallbackError::Permanent {
|
||||
message: "photo dimensions exceed Telegram limits and no smaller variant is available"
|
||||
.into(),
|
||||
})?;
|
||||
let media = media_from_url(&item, url, caption, item.thumbnail_url())
|
||||
.map_err(|message| FallbackError::Permanent { message })?;
|
||||
Ok(PreparedItem {
|
||||
index,
|
||||
media,
|
||||
keep_alive: None,
|
||||
})
|
||||
}
|
||||
}
|
||||
} else {
|
||||
let path = file.path().to_path_buf();
|
||||
let media = media_from_file(&item, path, caption, item.thumbnail_url())
|
||||
.map_err(|message| FallbackError::Permanent { message })?;
|
||||
Ok(PreparedItem {
|
||||
index,
|
||||
media,
|
||||
keep_alive: Some(file),
|
||||
})
|
||||
}
|
||||
}
|
||||
Err(FallbackError::MediaTooLarge) => {
|
||||
let url = item
|
||||
.fallback_url()
|
||||
.ok_or_else(|| FallbackError::Permanent {
|
||||
message: "media too large".into(),
|
||||
})?;
|
||||
let media = media_from_url(&item, url, caption, item.thumbnail_url())
|
||||
.map_err(|message| FallbackError::Permanent { message })?;
|
||||
Ok(PreparedItem {
|
||||
index,
|
||||
media,
|
||||
keep_alive: None,
|
||||
})
|
||||
}
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
}
|
||||
|
||||
/// Download-and-reupload fallback for one media batch. Files over the upload
|
||||
/// cap are not downloaded/uploaded; the item falls back to its smaller URL
|
||||
/// (which Telegram fetches itself). Items are prepared concurrently (bounded)
|
||||
/// because the downloads are network-bound; the batch is then uploaded in its
|
||||
/// original order. Returns the fallback-error without the task attached;
|
||||
/// callers wrap it with the updated task state.
|
||||
pub(super) async fn send_batch_via_upload(
|
||||
sender: &dyn MediaSender,
|
||||
chat_id: i64,
|
||||
reply_to: i64,
|
||||
batch: &[MediaItemPayload],
|
||||
caption: Option<&str>,
|
||||
task: Task,
|
||||
) -> Result<Vec<Message>, SendError> {
|
||||
let sem = std::sync::Arc::new(tokio::sync::Semaphore::new(3));
|
||||
let mut set = tokio::task::JoinSet::new();
|
||||
for (i, item) in batch.iter().enumerate() {
|
||||
let item_caption = if i == 0 {
|
||||
caption.map(str::to_string)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let item = item.clone();
|
||||
let sem = std::sync::Arc::clone(&sem);
|
||||
set.spawn(async move {
|
||||
let _permit = sem.acquire().await.expect("upload semaphore closed");
|
||||
prepare_upload_item(item, i, item_caption.as_deref()).await
|
||||
});
|
||||
}
|
||||
let mut prepared: Vec<Option<InputMedia>> = (0..batch.len()).map(|_| None).collect();
|
||||
let mut keep_alive: Vec<NamedTempFile> = Vec::new();
|
||||
while let Some(joined) = set.join_next().await {
|
||||
let item = match joined {
|
||||
Ok(Ok(item)) => item,
|
||||
// Dropping the JoinSet aborts the remaining prep tasks; their
|
||||
// temp files are cleaned up on drop (short-circuit like before).
|
||||
Ok(Err(e)) => return Err(SendError::from_fallback(e, task.clone())),
|
||||
Err(e) => {
|
||||
return Err(SendError::Permanent {
|
||||
message: format!("upload worker panicked: {e}"),
|
||||
task: Box::new(task),
|
||||
});
|
||||
}
|
||||
};
|
||||
let PreparedItem {
|
||||
index,
|
||||
media,
|
||||
keep_alive: file_opt,
|
||||
} = item;
|
||||
if let Some(file) = file_opt {
|
||||
keep_alive.push(file);
|
||||
}
|
||||
prepared[index] = Some(media);
|
||||
}
|
||||
let items: Vec<InputMedia> = prepared
|
||||
.into_iter()
|
||||
.map(|m| m.expect("every upload item was prepared"))
|
||||
.collect();
|
||||
// `keep_alive` holds the temp files until the group request completes.
|
||||
let result = sender
|
||||
.send_media_group(ChatId(chat_id), MessageId(reply_to as i32), items)
|
||||
.await;
|
||||
drop(keep_alive);
|
||||
match result {
|
||||
Ok(messages) => Ok(messages),
|
||||
Err(e) => Err(classify_to_send_error(&e, task, "upload failed")),
|
||||
}
|
||||
}
|
||||
+218
-79
@@ -1,12 +1,13 @@
|
||||
//! Per-chat state with SQLite persistence (table `chat_state` in
|
||||
//! `data/task_queue.db`, shared with the task queue).
|
||||
|
||||
use crate::db::unix_now;
|
||||
use parking_lot::Mutex;
|
||||
use rusqlite::params;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::HashMap;
|
||||
use std::path::Path;
|
||||
use std::time::{Duration, SystemTime, UNIX_EPOCH};
|
||||
use std::sync::Arc;
|
||||
use std::time::Duration;
|
||||
|
||||
#[derive(Serialize, Deserialize, Default, Clone, Debug)]
|
||||
pub struct ChatData {
|
||||
@@ -16,8 +17,8 @@ pub struct ChatData {
|
||||
pub edit_message: HashMap<i64, EditMessage>,
|
||||
/// name -> HTML template containing "[]"
|
||||
pub template: HashMap<String, String>,
|
||||
/// site name (twitter/bsky/pixiv) -> user-supplied caption format with
|
||||
/// {url} {author} {author_url} {title} {tags} placeholders.
|
||||
/// site name (twitter/bsky/misskey/pixiv/bilibili) -> user-supplied caption format
|
||||
/// with {url} {author} {author_url} {title} {content} {tags} placeholders.
|
||||
pub message_format: HashMap<String, String>,
|
||||
}
|
||||
|
||||
@@ -34,36 +35,22 @@ pub struct EditMessage {
|
||||
pub struct ChatStore {
|
||||
/// In-memory cache; the DB is the source of truth on first access.
|
||||
cache: Mutex<HashMap<i64, ChatData>>,
|
||||
db_path: String,
|
||||
}
|
||||
|
||||
pub fn unix_now() -> i64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.map(|d| d.as_secs() as i64)
|
||||
.unwrap_or(0)
|
||||
/// Per-chat async locks serializing get→mutate→set so concurrent handler
|
||||
/// tasks (batch-forwards, callbacks) cannot clobber each other's writes.
|
||||
locks: Mutex<HashMap<i64, Arc<tokio::sync::Mutex<()>>>>,
|
||||
pool: Arc<crate::db::DbPool>,
|
||||
}
|
||||
|
||||
impl ChatStore {
|
||||
/// Creates the parent directory and the `chat_state` table (idempotent).
|
||||
/// The shared `tasks` / `link_cache` tables are owned by `queue.rs` and
|
||||
/// `link_cache.rs` respectively.
|
||||
pub fn open(path: &str) -> rusqlite::Result<Self> {
|
||||
if let Some(parent) = Path::new(path).parent()
|
||||
&& !parent.as_os_str().is_empty()
|
||||
{
|
||||
std::fs::create_dir_all(parent)
|
||||
.map_err(|e| rusqlite::Error::ToSqlConversionFailure(Box::new(e)))?;
|
||||
}
|
||||
let conn = crate::db::open_db(path)?;
|
||||
conn.execute_batch(
|
||||
"CREATE TABLE IF NOT EXISTS chat_state (chat_id TEXT PRIMARY KEY, payload TEXT NOT NULL);",
|
||||
)?;
|
||||
drop(conn);
|
||||
Ok(ChatStore {
|
||||
/// Wraps the shared DB pool (schema initialized once by
|
||||
/// [`crate::db::open_store`]; the `chat_state` table lives in the merged
|
||||
/// schema alongside `tasks` and `link_cache`).
|
||||
pub fn new(pool: Arc<crate::db::DbPool>) -> Self {
|
||||
ChatStore {
|
||||
cache: Mutex::new(HashMap::new()),
|
||||
db_path: path.to_string(),
|
||||
})
|
||||
locks: Mutex::new(HashMap::new()),
|
||||
pool,
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn get(&self, chat_id: i64) -> ChatData {
|
||||
@@ -71,23 +58,25 @@ impl ChatStore {
|
||||
return data.clone();
|
||||
}
|
||||
let chat_key = chat_id.to_string();
|
||||
let payload = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
// Concurrent handler tasks (batch-forwards) may write chat_state
|
||||
// while this read runs; the shared busy timeout handles the
|
||||
// write-lock collision instead of failing the query.
|
||||
let mut stmt = conn.prepare("SELECT payload FROM chat_state WHERE chat_id = ?1")?;
|
||||
let mut rows = stmt.query(params![chat_key])?;
|
||||
match rows.next()? {
|
||||
Some(row) => Ok(Some(row.get::<_, String>(0)?)),
|
||||
None => Ok(None),
|
||||
}
|
||||
})
|
||||
.await
|
||||
.unwrap_or_else(|e| {
|
||||
log::error!("chat_state read failed: {e}");
|
||||
None
|
||||
})
|
||||
.unwrap_or_default();
|
||||
let payload = self
|
||||
.pool
|
||||
.with_conn(move |conn| {
|
||||
// Concurrent handler tasks (batch-forwards) may write chat_state
|
||||
// while this read runs; the shared busy timeout handles the
|
||||
// write-lock collision instead of failing the query.
|
||||
let mut stmt = conn.prepare("SELECT payload FROM chat_state WHERE chat_id = ?1")?;
|
||||
let mut rows = stmt.query(params![chat_key])?;
|
||||
match rows.next()? {
|
||||
Some(row) => Ok(Some(row.get::<_, String>(0)?)),
|
||||
None => Ok(None),
|
||||
}
|
||||
})
|
||||
.await
|
||||
.unwrap_or_else(|e| {
|
||||
log::error!("chat_state read failed: {e}");
|
||||
None
|
||||
})
|
||||
.unwrap_or_default();
|
||||
let data: ChatData = serde_json::from_str(&payload).unwrap_or_default();
|
||||
self.cache.lock().insert(chat_id, data.clone());
|
||||
data
|
||||
@@ -98,50 +87,97 @@ impl ChatStore {
|
||||
self.cache.lock().insert(chat_id, data.clone());
|
||||
let payload = serde_json::to_string(data).expect("chat state serializes");
|
||||
let chat_id = chat_id.to_string();
|
||||
let result = crate::db::with_conn(&self.db_path, move |conn| {
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO chat_state (chat_id, payload) VALUES (?1, ?2)",
|
||||
params![chat_id, payload],
|
||||
)?;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
let result = self
|
||||
.pool
|
||||
.with_conn(move |conn| {
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO chat_state (chat_id, payload) VALUES (?1, ?2)",
|
||||
params![chat_id, payload],
|
||||
)?;
|
||||
Ok(())
|
||||
})
|
||||
.await;
|
||||
if let Err(e) = result {
|
||||
log::error!("chat_state write failed: {e}");
|
||||
}
|
||||
}
|
||||
|
||||
/// The per-chat async lock serializing get→mutate→set cycles.
|
||||
fn lock_for(&self, chat_id: i64) -> Arc<tokio::sync::Mutex<()>> {
|
||||
self.locks
|
||||
.lock()
|
||||
.entry(chat_id)
|
||||
.or_insert_with(|| Arc::new(tokio::sync::Mutex::new(())))
|
||||
.clone()
|
||||
}
|
||||
|
||||
/// Serializes a get→mutate→set cycle per chat: concurrent handler tasks
|
||||
/// (the batch-forward design spawns several per chat) each snapshot the
|
||||
/// same `ChatData` and last-writer-wins would silently drop mutations,
|
||||
/// e.g. a second `edit_message` record. The per-chat lock makes the
|
||||
/// cycle atomic. Returns the closure's result.
|
||||
pub async fn update<R>(&self, chat_id: i64, f: impl FnOnce(&mut ChatData) -> R) -> R {
|
||||
let lock = self.lock_for(chat_id);
|
||||
let _guard = lock.lock().await;
|
||||
let mut data = self.get(chat_id).await;
|
||||
let r = f(&mut data);
|
||||
self.set(chat_id, &data).await;
|
||||
r
|
||||
}
|
||||
|
||||
/// Removes edit-before-forward records whose `created_at + ttl` is in the
|
||||
/// past. Returns the removed `(chat_id, prompt_message_id)` pairs so the
|
||||
/// caller can clear the prompt's buttons.
|
||||
pub async fn prune_expired(&self, ttl: Duration) -> Vec<(i64, i64)> {
|
||||
let now = unix_now();
|
||||
let ttl_secs = ttl.as_secs() as i64;
|
||||
let mut removed = Vec::new();
|
||||
let changed: Vec<(i64, ChatData)> = {
|
||||
let mut cache = self.cache.lock();
|
||||
let mut out = Vec::new();
|
||||
for (chat_id, data) in cache.iter_mut() {
|
||||
let keys: Vec<i64> = data.edit_message.keys().copied().collect();
|
||||
let mut kept = HashMap::new();
|
||||
for key in keys {
|
||||
if let Some(entry) = data.edit_message.get(&key) {
|
||||
if entry.created_at + ttl_secs > now {
|
||||
kept.insert(key, entry.clone());
|
||||
} else {
|
||||
removed.push((*chat_id, key));
|
||||
}
|
||||
}
|
||||
}
|
||||
if kept.len() != data.edit_message.len() {
|
||||
data.edit_message = kept;
|
||||
out.push((*chat_id, data.clone()));
|
||||
}
|
||||
}
|
||||
out
|
||||
// Chats that may have an expired record, from a cache snapshot; the
|
||||
// pruning itself re-reads and writes under the per-chat lock below
|
||||
// (see the eviction note). Takes no lock of its own, so a chat
|
||||
// appearing later is simply picked up by the next sweep.
|
||||
let candidates: Vec<i64> = {
|
||||
let cache = self.cache.lock();
|
||||
cache
|
||||
.iter()
|
||||
.filter(|(_, data)| {
|
||||
data.edit_message
|
||||
.values()
|
||||
.any(|entry| entry.created_at + ttl_secs <= now)
|
||||
})
|
||||
.map(|(chat_id, _)| *chat_id)
|
||||
.collect()
|
||||
};
|
||||
for (chat_id, data) in changed {
|
||||
self.set(chat_id, &data).await;
|
||||
let mut removed = Vec::new();
|
||||
let mut evicted_chats = Vec::new();
|
||||
for chat_id in candidates {
|
||||
let lock = self.lock_for(chat_id);
|
||||
let _guard = lock.lock().await;
|
||||
let mut data = self.get(chat_id).await;
|
||||
let before = data.edit_message.len();
|
||||
data.edit_message.retain(|key, entry| {
|
||||
if entry.created_at + ttl_secs > now {
|
||||
return true;
|
||||
}
|
||||
removed.push((chat_id, *key));
|
||||
false
|
||||
});
|
||||
if data.edit_message.len() != before {
|
||||
self.set(chat_id, &data).await;
|
||||
}
|
||||
// Chats with no live edit records: evicted from the cache (and
|
||||
// their per-chat lock) so the cache stays bounded to active
|
||||
// prompts. The DB keeps the row; the next get() reloads it.
|
||||
if data.edit_message.is_empty() {
|
||||
evicted_chats.push(chat_id);
|
||||
}
|
||||
}
|
||||
if !evicted_chats.is_empty() {
|
||||
let mut cache = self.cache.lock();
|
||||
let mut locks = self.locks.lock();
|
||||
for chat_id in &evicted_chats {
|
||||
cache.remove(chat_id);
|
||||
locks.remove(chat_id);
|
||||
}
|
||||
}
|
||||
if !removed.is_empty() {
|
||||
log::info!(
|
||||
@@ -152,3 +188,106 @@ impl ChatStore {
|
||||
removed
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[tokio::test]
|
||||
async fn concurrent_updates_do_not_lose_edit_records() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pool = crate::db::open_store(dir.path().join("s.db").to_str().unwrap()).unwrap();
|
||||
let store = std::sync::Arc::new(ChatStore::new(pool));
|
||||
let mut handles = Vec::new();
|
||||
for i in 0..4 {
|
||||
let store = Arc::clone(&store);
|
||||
handles.push(tokio::spawn(async move {
|
||||
store
|
||||
.update(1001, |data| {
|
||||
data.edit_message.insert(
|
||||
i,
|
||||
EditMessage {
|
||||
url: format!("https://x.com/u/status/{i}"),
|
||||
chat_id: 1001,
|
||||
forward_message_ids: vec![i],
|
||||
template: String::new(),
|
||||
created_at: 0,
|
||||
},
|
||||
);
|
||||
})
|
||||
.await;
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.await.unwrap();
|
||||
}
|
||||
let data = store.get(1001).await;
|
||||
assert_eq!(
|
||||
data.edit_message.len(),
|
||||
4,
|
||||
"concurrent get→mutate→set must not drop records"
|
||||
);
|
||||
}
|
||||
|
||||
fn edit_entry(chat_id: i64, created_at: i64) -> EditMessage {
|
||||
EditMessage {
|
||||
url: "https://x.com/u/status/1".into(),
|
||||
chat_id,
|
||||
forward_message_ids: vec![9],
|
||||
template: String::new(),
|
||||
created_at,
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn prune_removes_only_expired_records() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pool = crate::db::open_store(dir.path().join("p.db").to_str().unwrap()).unwrap();
|
||||
let store = ChatStore::new(pool);
|
||||
let now = unix_now();
|
||||
store
|
||||
.update(7, |data| {
|
||||
data.template.insert("t".into(), "[]".into());
|
||||
data.edit_message.insert(1, edit_entry(7, now - 3600));
|
||||
data.edit_message.insert(2, edit_entry(7, now));
|
||||
})
|
||||
.await;
|
||||
|
||||
let removed = store.prune_expired(Duration::from_secs(60)).await;
|
||||
|
||||
assert_eq!(removed, vec![(7, 1)]);
|
||||
let data = store.get(7).await;
|
||||
assert!(data.edit_message.contains_key(&2), "live record pruned");
|
||||
assert_eq!(
|
||||
data.template.get("t").map(String::as_str),
|
||||
Some("[]"),
|
||||
"unrelated state lost by the prune"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn prune_eviction_keeps_the_persisted_state() {
|
||||
// Every record expires → the chat is evicted from the cache; the
|
||||
// pruned state must already be in the DB when that happens.
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pool = crate::db::open_store(dir.path().join("p.db").to_str().unwrap()).unwrap();
|
||||
let store = ChatStore::new(pool);
|
||||
store
|
||||
.update(8, |data| {
|
||||
data.template.insert("keep".into(), "[]".into());
|
||||
data.edit_message.insert(1, edit_entry(8, 0));
|
||||
})
|
||||
.await;
|
||||
|
||||
let removed = store.prune_expired(Duration::from_secs(60)).await;
|
||||
|
||||
assert_eq!(removed, vec![(8, 1)]);
|
||||
let data = store.get(8).await;
|
||||
assert!(data.edit_message.is_empty());
|
||||
assert_eq!(
|
||||
data.template.get("keep").map(String::as_str),
|
||||
Some("[]"),
|
||||
"eviction dropped state the DB never received"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -55,6 +55,14 @@ services:
|
||||
depends_on:
|
||||
- nginx-proxy
|
||||
container_name: tgxmb
|
||||
# Webhook mode only: the bot listens on WEBHOOK_PORT; nginx-proxy shows
|
||||
# 502s while this is down, so surface it to the orchestrator.
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/127.0.0.1/8443'"]
|
||||
interval: 30s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 10s
|
||||
|
||||
volumes:
|
||||
certs:
|
||||
|
||||
@@ -16,7 +16,6 @@ then
|
||||
else
|
||||
usermod -u ${USER_ID} -o user > /dev/null 2>&1 || true
|
||||
fi
|
||||
usermod -a -G root user > /dev/null 2>&1 || true
|
||||
# Bind-mounted volumes may not support chown; a failure here must not kill
|
||||
# the container either.
|
||||
chown -R `id -u user`:`id -u user` /app > /dev/null 2>&1 || true
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
# 架构优化设计:可测试性接缝 + handlers 拆分
|
||||
|
||||
> 状态:**阶段 A、B、C 已实施**(A: `c9e72fd`,B: `50206a9` + `ae69d72`,C:
|
||||
> rate_limit 提交);**D 已延迟**——待下次数据库 schema 变化时实施(见 §5)。
|
||||
> 目标:把仓库最大的测试空白(`handlers.rs`/`send.rs` 的发送与分派逻辑)补上
|
||||
> 可测试接缝,并把 ~1100 行的 handlers 单体拆成模块。
|
||||
|
||||
---
|
||||
|
||||
## 1. 现状与动机
|
||||
|
||||
- `handlers.rs`(~1100 行)混装:命令解析/执行、URL 提取 + 任务通道、inline
|
||||
debounce、callback、edit-before-forward、全部全局静态。
|
||||
- 关键路径零测试:`url_media` 的分派、`dispatch_send` 的失败分类、缓存命中路径、
|
||||
edit-before-forward、转发重试——AGENTS.md 自认 "untested: handlers.rs"。
|
||||
- 根因:`handlers.rs`/`send.rs` 直接依赖 teloxide `Bot`(具体类型)与全局静态
|
||||
(`CHAT_STORE`/`TASK_QUEUE`/`LINK_CACHE`/`CONFIG`),没有注入点。
|
||||
|
||||
## 2. 阶段 A:handlers 拆分(纯组织,零风险,先行)
|
||||
|
||||
把 `handlers.rs` 拆为模块(仅移动代码,不改签名):
|
||||
|
||||
```
|
||||
handlers/
|
||||
mod.rs — 入口:message/inline/callback 分发 + 公共类型(UrlJob、log_key)
|
||||
statics.rs — CHAT_STORE / TASK_QUEUE / LINK_CACHE / DB / CONFIG / URL_JOBS
|
||||
commands.rs — Command enum + execute_command + set_forward_channel_handler
|
||||
urls.rs — extract_urls + start/stop_url_workers + url_media + build_send_task + media_to_payload
|
||||
inline.rs — inline_query_handler + debounce 状态机 + answer_inline_query
|
||||
callback.rs — callback_query_handler + edit_message_handler
|
||||
```
|
||||
|
||||
- `mod.rs` 用 `pub use` 重导出,bot 侧引用 `handlers::xxx` 不变。
|
||||
- 收益:每个模块独立审阅;后续阶段 B 的接缝改动落在明确的模块内。
|
||||
|
||||
## 3. 阶段 B:MediaSender 接缝(核心)
|
||||
|
||||
**动机**:`send.rs` 的所有发送入口(`send_media_group`/`send_animation`/
|
||||
`copy_messages`)都挂在具体 `Bot` 上;测试无法注入失败/成功。
|
||||
|
||||
**设计**:新增 `crates/xmedia-bot/src/media_sender.rs`:
|
||||
|
||||
```rust
|
||||
/// 发送抽象:生产用 teloxide Bot,测试用记录型 mock。
|
||||
/// 方法签名与 teloxide 调用点一一对应,返回 Result 以便注入任意失败。
|
||||
pub trait MediaSender: Send + Sync {
|
||||
fn send_media_group(&self, chat_id: ChatId, items: Vec<InputMedia>)
|
||||
-> BoxFuture<'_, Result<Vec<Message>, RequestError>>;
|
||||
fn send_animation(&self, chat_id: ChatId, file: InputFile, caption: Option<&str>, spoiler: bool, reply_to: i64)
|
||||
-> BoxFuture<'_, Result<Message, RequestError>>;
|
||||
fn copy_messages(&self, to: ChatId, from: ChatId, ids: Vec<MessageId>)
|
||||
-> BoxFuture<'_, Result<Vec<MessageId>, RequestError>>;
|
||||
// 按需扩展:edit_message_caption / delete_message / answer_callback_query …
|
||||
}
|
||||
|
||||
impl MediaSender for Bot { /* 委托现有 teloxide 调用 */ }
|
||||
```
|
||||
|
||||
配套:`ChatStore`/`LinkCache`/`PersistentTaskQueue` 已是具体类型——给 `send.rs`/
|
||||
`url_media` 需要的最小面加 trait(`ChatStoreReader`/`LinkCacheReader` 等),或直接
|
||||
注入具体类型(它们已有内存态,测试用真实 tempdir 即可,见阶段 B-注)。
|
||||
|
||||
**接入点**:
|
||||
- `dispatch_send` / `send_media_sequence` / `send_animation` / `forward_messages` /
|
||||
`post_send_actions` / `notify_failure` 的 `bot: &Bot` 参数改为 `sender: &dyn MediaSender`。
|
||||
- `url_media` 由 `url_media(bot, message, url)` 改为 `url_media(sender, store, queue, cache, message, url)`(或聚合为一个 `AppContext` 结构传引用)。
|
||||
|
||||
**测试策略**(仓库无 mock 框架,手写 mock):
|
||||
- `MockSender` 记录调用序列、按脚本返回 Ok/Err(覆盖:URL 发送成功、media-fetch
|
||||
失败触发兜底、RetryAfter 触发入队、Permanent 触发缓存失效)。
|
||||
- `ChatStore`/`LinkCache` 用真实 tempdir 实例(现有测试已这么做)。
|
||||
- 新增测试:`send_media_sequence` 分批续传、`send_animation` 兜底、`url_media`
|
||||
缓存命中 vs 未命中、`dispatch_send` 三分支。
|
||||
|
||||
**风险**:中。动 `send.rs`/`handlers.rs` 签名(约 15 处调用点),行为不变。
|
||||
**不做**:`main.rs` 的 teloxide 装配不抽象(那是真正的胶水,无测试价值)。
|
||||
|
||||
## 4. 阶段 C:主动限流(已实施)
|
||||
|
||||
批量转发时的突发会触发 Telegram 频道限速,现在靠 `RetryAfter → 队列重试` 被动
|
||||
应对。新增轻量令牌桶(`rate_limit.rs`):
|
||||
|
||||
```rust
|
||||
pub struct TokenBucket { capacity, refill_per_sec, state: Mutex<State> }
|
||||
impl TokenBucket {
|
||||
pub async fn acquire(&self, n: f64); // 按 n 个 token 等待并消费
|
||||
}
|
||||
pub fn limiter_for(chat_id: i64) -> Arc<TokenBucket>; // 每频道一个桶
|
||||
```
|
||||
|
||||
- 默认 `CAPACITY = 20`、`REFILL_PER_SEC = 20/60`(约 20 msg/min);
|
||||
单次 acquire 可超出容量(记为债务,由后续 refill 偿还)。
|
||||
- 挂点:`MediaSender for Bot` 的 `send_media_group`(按 items 数)、
|
||||
`copy_messages`(按 ids 数)、`send_animation`(1 token)前置 `acquire`;
|
||||
MockSender 不受影响(测试不经过限流)。
|
||||
- 收益:减少 429 → 重试 → 死信;队列重试仍是全局限速的安全网。
|
||||
- 风险:低,独立模块;`tokio::time`(paused-clock 可测)。
|
||||
|
||||
## 5. 阶段 D:DB 版本化迁移(**已延迟**)
|
||||
|
||||
> ⚠️ **待办提醒**:本阶段**推迟到下次数据库 schema 变化时实施**(给
|
||||
> `link_cache`/`chat_state`/`tasks` 加列、改结构等)。当前 `schema_init` 是
|
||||
> `CREATE TABLE IF NOT EXISTS`,无版本概念;一旦需要迁移已有线上库,必须先落地
|
||||
> 本方案(`PRAGMA user_version` 迁移链)再改 schema。`db.rs` 的 `schema_init`
|
||||
> 处已留注释指向这里。
|
||||
|
||||
```rust
|
||||
// db.rs
|
||||
const MIGRATIONS: &[&str] = &[
|
||||
// v1: 初始 schema(tasks / chat_state / link_cache)
|
||||
"CREATE TABLE IF NOT EXISTS tasks (...); ...",
|
||||
];
|
||||
pub fn migrate(conn: &Connection) -> rusqlite::Result<()> {
|
||||
let v: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
|
||||
for (i, sql) in MIGRATIONS.iter().enumerate().skip(v as usize) {
|
||||
conn.execute_batch(sql)?;
|
||||
conn.pragma_update(None, "user_version", (i + 1) as i64)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
```
|
||||
|
||||
- 低优先级:schema 未变时无收益;将来加列/改结构时必须有。
|
||||
- `open_store` 改用 `migrate` 替换 `schema_init` 调用。
|
||||
|
||||
## 6. 明确不做
|
||||
|
||||
- **不拆 xmedia-core**:`Task`/队列/发送抽成独立 lib crate 是大工程,除非出现
|
||||
第二个客户端,否则收益不抵成本。
|
||||
- **不引入 DI 框架**:仓库惯例是 LazyLock 静态 + 显式传参,保持。
|
||||
- **不抽象 main.rs 的 teloxide 装配**。
|
||||
|
||||
## 7. 实施记录
|
||||
|
||||
| 阶段 | 提交 | 说明 |
|
||||
|---|---|---|
|
||||
| A | `c9e72fd` | handlers 拆为 `{mod, statics, commands, urls, inline, callback}` |
|
||||
| B | `50206a9` | `media_sender.rs`:`trait MediaSender` + `impl for Bot`(`<Bot as Requester>::` 消歧);send.rs 8 处签名改 `&dyn MediaSender`;`MockSender` 测试覆盖兜底触发与错误分类(+5 测试) |
|
||||
| B | `ae69d72` | `AppContext` 注入 `url_media`(sender/store/queue/cache),url_media 全链路测试(缓存命中/失效/成功/不支持 URL,+3 测试) |
|
||||
| C | rate_limit 提交 | `rate_limit.rs` 令牌桶 + 每频道注册表;`MediaSender for Bot` 的 group/copy/animation 前置 `acquire`(+3 测试) |
|
||||
| D | — | **已延迟**:待下次数据库 schema 变化时实施(见 §5) |
|
||||
|
||||
A、B、C 为核心并已实施;D 在 schema 变更时落地。
|
||||
@@ -0,0 +1,241 @@
|
||||
# 站点适配器重构方案:让新增站点变成"新模块 + 注册一行"
|
||||
|
||||
> 状态:**已实施**(阶段 1-5,提交 `7ca8fd1` / `5e23916` / `bf4e615` / `5679a8c` +
|
||||
> 本文档收尾)。目标:把"加一个新站点"从改 8-9 处收敛到 3 处,并让站点身份、
|
||||
> 重试策略、下载 header 等站点能力归位到站点模块自身。实施过程中的关键偏差
|
||||
> (async 形态)见 §3 的 "async 形态" 段——原生 AFIT 实测不可用于 dyn 分派,
|
||||
> 最终采用手写 `BoxFuture`(`SiteFuture` 别名)。
|
||||
|
||||
---
|
||||
|
||||
## 1. 现状摩擦清单
|
||||
|
||||
> ⚠️ 本节记录的是**重构前**的现状:其中的行号、以及 `site/mod.rs` 里的
|
||||
> `fetch_once` 分派函数(当时的实现)都已不存在,仅作历史记录。当前形态见
|
||||
> `site/mod.rs` 的 `SITES` 注册表——新增站点 = 新模块 + 注册一行。
|
||||
|
||||
以现有三站(twitter / bsky / pixiv)为基线,新增第 4 个站点(代号 `example`)
|
||||
今天需要触碰的位置:
|
||||
|
||||
| # | 位置(当前行号) | 改动 | 必改? |
|
||||
|---|---|---|---|
|
||||
| 1 | 新目录 `crates/x-media/src/site/example/{mod,interface,model}.rs` | 新模块 | 必改 |
|
||||
| 2 | `site/mod.rs:354-365` `fetch_once` | 加一个 `if` 分派分支 | 必改 |
|
||||
| 3 | `site/mod.rs:167-178` `cache_key` | 加一个 `if` 分支 + 约定 key 前缀 `"example:..."` | 必改 |
|
||||
| 4 | `site/mod.rs:53-63` `site_name()` | 加一个 URL `contains` 嗅探分支 | 必改 |
|
||||
| 5 | `handlers.rs:405` `SetFormat` 白名单 | `["twitter","bsky","pixiv"]` 加字符串 | 必改 |
|
||||
| 6 | `config.rs` / `main.rs:74-84` | 仿 pixiv 加启动校验(token、`disable()`) | 视站点 |
|
||||
| 7 | `site/mod.rs:312-326` `fetch_error_is_retryable` | 若重试策略特殊,改中央分类函数 | 视站点 |
|
||||
| 8 | `site/mod.rs:377/391/430` 三个下载函数 | 若媒体有防盗链,加 header(现在是硬编码 pximg 判断) | 视站点 |
|
||||
| 9 | `site/mod.rs:16,180-197` `FetchError` | 若错误类型特殊,加嵌套 variant(仿 `Pixiv(PixivError)`) | 视站点 |
|
||||
|
||||
**根因**:仓库里没有"站点"这个实体。站点的四类能力——URL 识别(PATTERN +
|
||||
cache_key)、抓取、重试策略、下载 header——分别散落在中央 if 链、URL 字符串嗅探、
|
||||
魔法字符串 key 和 bot crate 的白名单里。`AGENTS.md` 现行约定 "no trait, no enum
|
||||
dispatch" 是刻意的简单性选择;本方案的目标是在**不推翻它精神的前提下**收敛摩擦,
|
||||
并在阶段 3 提供完整的 trait 注册表选项。
|
||||
|
||||
## 2. 目标架构
|
||||
|
||||
```
|
||||
crates/x-media/src/site/mod.rs
|
||||
├─ SITES: LazyLock<Vec<Box<dyn Site>>> ← 注册表(唯一的"站点列表")
|
||||
├─ find_site(url) / fetch(url) / cache_key(url) / site_ids()
|
||||
└─ 通用类型:Fetched { site_id, ... } / FetchError(通用类 + Site 变体)
|
||||
│
|
||||
├─ site/twitter/{mod,interface,model}.rs impl Site
|
||||
├─ site/bsky/… impl Site
|
||||
└─ site/pixiv/… impl Site (download_headers: pximg Referer)
|
||||
(validate: token 校验)
|
||||
|
||||
crates/xmedia-bot
|
||||
├─ handlers.rs SetFormat 白名单 ← x_media::site::ids()(不再写死)
|
||||
├─ handlers.rs site 格式查找 ← fetched.site_id(缓存/新鲜两条路径同口径)
|
||||
└─ main.rs 启动校验 ← site::validate_all()(不再特判 pixiv)
|
||||
```
|
||||
|
||||
## 3. 分阶段迁移
|
||||
|
||||
每个阶段是一个独立提交,保持 `cargo fmt` / `cargo clippy -- -D warnings` /
|
||||
`cargo test --workspace` 全绿;行为完全不变,只挪代码、不换语义。
|
||||
|
||||
### 阶段 1:站点身份单一来源(低风险,推荐先做)
|
||||
|
||||
**动机**:同一概念目前有两个来源——缓存命中路径用 `key.split(':').next()`
|
||||
(`handlers.rs:639`),新鲜抓取路径用 `fetched.site_name()`(`handlers.rs:724`);
|
||||
`site_name()` 又是对 `source_url` 的 `contains` 字符串嗅探,还有 `"unknown"`
|
||||
兜底分支。
|
||||
|
||||
**改动**:
|
||||
|
||||
1. `site/mod.rs`:`Fetched` 增加字段 `site_id: &'static str`(由各站点的
|
||||
`impl From<SiteStruct> for Fetched` 填充;`empty_fetched` 同步填)。
|
||||
`Fetched::site_name()` 改为 `return self.site_id`(保留方法名,删除
|
||||
`source_url.contains` 嗅探与 `"unknown"` 分支)。
|
||||
2. `site/mod.rs`:新增 `pub fn site_id_from_key(key: &str) -> &'static str`
|
||||
(解析 `"example:..."` 前缀,未知前缀返回 `"unknown"`),bot 缓存命中路径改用它,
|
||||
与 `fetched.site_id` 口径统一。
|
||||
3. `handlers.rs:405`:`SetFormat` 白名单改为 `x_media::site::ids()`——阶段 1 先实现
|
||||
`ids()` 为 `["twitter","bsky","pixiv"]` 的常量函数(数据源仍集中,行为不变),
|
||||
阶段 3 再改为遍历注册表。
|
||||
4. `twitter/interface.rs:48-60` / `bsky` / `pixiv` 的 `From<SiteStruct> for Fetched`
|
||||
各补 `site_id` 字段。
|
||||
|
||||
**风险**:低。纯增量字段;`site_name()` 语义不变(测试 `pixiv/interface.rs:355`
|
||||
已断言 `"pixiv"`)。
|
||||
**验证**:现有全部单测;`cache_key_normalizes_domain_variants` 等不变。
|
||||
**回滚**:revert 该提交。
|
||||
|
||||
### 阶段 2:站点能力下沉(不引入 trait,静态分派)
|
||||
|
||||
**动机**:把"每个站点自己才知道"的逻辑搬回站点模块,中央只做迭代。这是
|
||||
`AGENTS.md` 现有约定(无 trait)与完整注册表之间的折中,可独立交付。
|
||||
|
||||
**改动**:每个站点模块新增并 `mod.rs` 重新导出:
|
||||
|
||||
```rust
|
||||
// site/twitter/interface.rs(bsky/pixiv 同构)
|
||||
pub fn cache_key(url: &str) -> Option<String>; // 用自身 PATTERN,返回 "twitter:<id>"
|
||||
pub fn is_retryable(err: &FetchError) -> bool; // 默认 Http|Transient;pixiv 覆盖 PixivError 分支
|
||||
pub fn media_headers(url: &str) -> Option<Vec<(&'static str, String)>>;
|
||||
// pixiv: url 含 "pximg.net" → Referer
|
||||
```
|
||||
|
||||
`site/mod.rs` 相应改为迭代三站:
|
||||
|
||||
- `cache_key`:逐个调 `site::cache_key`,不再自己写 key 格式;
|
||||
- `fetch_error_is_retryable`:删除,`fetch()` 重试循环改调 `current_site::is_retryable`
|
||||
(`fetch_once` 已能确定站点,把站点传下去);
|
||||
- `media_size` / `download_media_limited` / `download_media_to_file` 里的
|
||||
`pximg.net → Referer` 硬编码删除,改为遍历 `SITES`(阶段 2 是遍历
|
||||
`[twitter, bsky, pixiv]` 静态列表)取 `media_headers(url)` 合并。
|
||||
|
||||
**注意**:Referer 判定依据是媒体 URL 的 host(`pximg.net`),**不是**站点
|
||||
PATTERN(pixiv 的 PATTERN 只匹配 `pixiv.net/artworks/...`),所以 `media_headers`
|
||||
不能挂在 PATTERN 匹配上,必须按 URL 独立匹配——这正是把它做成独立函数的原因。
|
||||
|
||||
**风险**:中。下载函数签名不变,行为必须逐字节不变;新增单元测试覆盖
|
||||
`media_headers("https://i.pximg.net/...") == Some(Referer)` 与
|
||||
`cache_key` 等价性(对全部既有用例断言新旧结果一致)。
|
||||
**回滚**:revert。
|
||||
|
||||
### 阶段 3:Site trait + SITES 注册表(完整方案,可选)
|
||||
|
||||
**动机**:加站点时 bot crate 与中央分派零改动;站点列表成为唯一注册点。
|
||||
|
||||
**新增**(`site/mod.rs`,按实施后的实际形态):
|
||||
|
||||
```rust
|
||||
/// Boxed, Send future produced by a Site async method. Boxed so the trait
|
||||
/// stays dyn-compatible; Send because URL/queue workers tokio::spawn these.
|
||||
type SiteFuture<'a, T, E = FetchError> =
|
||||
Pin<Box<dyn Future<Output = Result<T, E>> + Send + 'a>>;
|
||||
|
||||
pub trait Site: Send + Sync {
|
||||
fn id(&self) -> &'static str;
|
||||
fn pattern(&self) -> &'static Regex;
|
||||
fn enabled(&self) -> bool { true } // 默认: true
|
||||
fn cache_key(&self, url: &str) -> Option<String>;
|
||||
fn fetch_from_url<'a>(&'a self, url: &'a str) -> SiteFuture<'a, Fetched>;
|
||||
fn is_retryable(&self, err: &FetchError) -> bool; // 默认: Http|Transient
|
||||
fn media_headers(&self, url: &str) -> Option<Vec<(&'static str, String)>>; // 默认: None
|
||||
fn validate(&self) -> SiteFuture<'static, (), String>; // 默认: Ok(())
|
||||
}
|
||||
|
||||
static SITES: LazyLock<Vec<Box<dyn Site>>> = LazyLock::new(|| vec![
|
||||
Box::new(twitter::TwitterSite), Box::new(bsky::BskySite), Box::new(pixiv::PixivSite),
|
||||
]);
|
||||
```
|
||||
|
||||
- `fetch` → `find_site(url)`(注册表中首个 PATTERN 命中且 `enabled()` 的站点,
|
||||
返回 `&'static dyn Site`)→ `site.fetch_from_url(url).await`;
|
||||
- `cache_key` / `site_ids()` / `site_id_from_key()` / `apply_media_headers()` /
|
||||
`validate_all()` 全部遍历 `SITES`;`validate_all` 返回失败列表,pixiv 的
|
||||
`Site::validate` 失败时自行 `disable()`;
|
||||
- `match_site`/`SiteKind`(阶段 2 的静态分派)与中央 `fetch_error_is_retryable`
|
||||
删除,重试判定走 `site.is_retryable`;
|
||||
- `main.rs` 的 pixiv 特判 → `site::validate_all()` + 通用失败通知;
|
||||
- 保留各站点的 `PATTERN`/`enabled()`/`fetch_from_url()` 顶层导出(兼容既有
|
||||
测试),trait impl 只是薄壳。
|
||||
|
||||
**async 形态**(实施结论):**原生 AFIT 不可行**。
|
||||
|
||||
- 实测(rustc 1.95.0,edition 2024;**1.97.1 复测一致**):trait 里写
|
||||
`async fn` 报 "method is `async`"(非 dyn 兼容);写反糖
|
||||
`-> impl Future<...> + Send + '_` 报 "references an `impl Trait` type in its
|
||||
return type"(同样非 dyn 兼容);纯 RPITIT(无 `+ Send`)也一样。即:
|
||||
**RPITIT/AFIT 目前无法用于 `Vec<Box<dyn Site>>` 注册表**,与早期设计的
|
||||
判断相反。
|
||||
- **为什么**:dyn 分派要求调用方在编译期知道返回值大小以分配空间,而
|
||||
`async fn`/RPITIT 返回不透明的 Future——这是"非定长返回值走 dyn"的普遍问题,
|
||||
与 async 无关。Rust 1.75 稳定的 AFIT 只覆盖**静态分派**,dyn 路径被排除;
|
||||
原生 dyn 支持(AFIDT)是 2026-2027 的已接受项目目标,尚未进入 stable。
|
||||
参见 <https://rust-lang.github.io/rust-project-goals/2026/afidt-box.html>。
|
||||
- **采用 (a) 手写 `Pin<Box<dyn Future + Send + '_>>`**(`SiteFuture` 别名):
|
||||
零新依赖、dyn 兼容、future 保证 Send。签名噪音靠别名缓解;生命周期坑因
|
||||
站点是无状态单元结构体 + `'a` 同时约束 `&self` 与 `url` 而完全可控
|
||||
(future 只借用调用域内的 url)。
|
||||
- **(b) `async-trait`** 仍是可行备选(语法更干净、同样 box),但新增依赖;
|
||||
本仓库采用 (a) 后无需引入。
|
||||
- 若未来 Rust 稳定版落地 AFIDT(调用点 `dyn_box!`),可平滑迁移回原生
|
||||
`async fn`,实现体几乎不动。
|
||||
|
||||
**风险**:中。动中央分派,但每站点行为不变;注册表迭代 + `find_site` 补单测
|
||||
(`fetch`/`cache_key` 对既有 URL 集合的结果与阶段 2 完全一致)。
|
||||
**回滚**:revert。
|
||||
|
||||
### 阶段 4:FetchError 泛化(已实施)
|
||||
|
||||
**改动**:`FetchError` 新增 `Site { site: &'static str, error: Box<dyn std::error::Error + Send + Sync> }`
|
||||
变体(`Display`/`source()` 同步)。**`Pixiv(PixivError)` 变体保留**(未迁移)——
|
||||
它已有完整的 `Display`/`source()`/`is_retryable` 处理,替换纯属 churn。`Site`
|
||||
变体默认永久性(各站点 `is_retryable` 都不匹配它);需要可重试站点错误的站点
|
||||
应自行转换为 `Http`/`Transient` 再返回。
|
||||
|
||||
**风险**:低(纯增量变体)。测试:`site_error_variant_displays_and_sources`。
|
||||
|
||||
### 阶段 5:收尾
|
||||
|
||||
- 更新 `AGENTS.md` 的 "Site adapter convention" 段:写新约定(注册表 + `impl Site` +
|
||||
每站点 `cache_key`/`is_retryable`/`media_headers`),删除 "no trait" 表述;
|
||||
- `examples/fetch.rs` 不变(走 `site::fetch`);
|
||||
- 新增站点 checklist 见 §4。
|
||||
|
||||
## 4. 重构后新增站点 checklist
|
||||
|
||||
```
|
||||
1. crates/x-media/src/site/example/{mod,interface,model}.rs // 新模块
|
||||
2. impl Site for ExampleSite 并注册进 SITES // 注册一行
|
||||
3. (可选)token 读取 + validate() 实现 // 启动校验自动生效
|
||||
── bot crate 零改动 ──
|
||||
```
|
||||
|
||||
对比现状的 8-9 处,bot crate 完全不碰:`SetFormat` 白名单、格式查找口径、
|
||||
缓存 key、启动校验全部自动跟随注册表。
|
||||
|
||||
## 5. 权衡与明确不做的事
|
||||
|
||||
- **不做**:Media 类型扩展(`media.rs` + `MediaItemPayload` + `CachedMediaKind` +
|
||||
send.rs 约 10+ 处 match 的 blast radius)——这是"新增媒体类型"的摩擦,与"新增
|
||||
站点"正交,优先级低,保持现状。
|
||||
- **不做**:DI/全局注入改造(`CHAT_STORE`/`TASK_QUEUE`/`CONFIG` 的 `LazyLock` 静态
|
||||
模式是仓库惯例,与站点扩展无关)。
|
||||
- **不做**:schema 迁移——新站点只产生新的 cache key 前缀与 `message_format` JSON
|
||||
key,`link_cache`/`chat_state` 表结构均无需变化。
|
||||
- **代价**:阶段 3 引入 `dyn Site` 与 boxed future 签名(`SiteFuture`,见 §3);
|
||||
`Send` 约束前移到 trait 边界,站点 impl 的 future 必须 Send(现仅在各
|
||||
`tokio::spawn` 点检查,重构后在 impl 处即报错,提前暴露问题)。
|
||||
若站点数量长期 ≤5 且无新增迹象,阶段 2 的折中方案已够用;本次已按完整方案
|
||||
实施到阶段 4。
|
||||
|
||||
## 6. 提交序列(已按此实施)
|
||||
|
||||
| 阶段 | 提交 | hash |
|
||||
|---|---|---|
|
||||
| 1 | `refactor(site): carry site_id on Fetched; unify cache-key site lookup` | `7ca8fd1` |
|
||||
| 2 | `refactor(site): move cache_key/is_retryable/media_headers into site modules` | `5e23916` |
|
||||
| 3 | `refactor(site): introduce Site trait and SITES registry` | `bf4e615` |
|
||||
| 4 | `refactor(site): genericize FetchError::Site` | `5679a8c` |
|
||||
| 5 | `docs: update site adapter convention in AGENTS.md` | 本文档收尾提交 |
|
||||
|
||||
每阶段独立合入、独立回滚;阶段 2 完成后"加站点"摩擦已收敛,3/4 为深化。
|
||||
Reference in New Issue
Block a user