Compare commits

..
1 Commits
Author SHA1 Message Date
YoursFunny 817b44c7bb use cached property 2024-06-20 02:27:24 +08:00
14 changed files with 148 additions and 390 deletions
+1 -1
View File
@@ -3,4 +3,4 @@ __pycache__/
cert/
data/
docker-compose.yml
utils/x.py
x.py
+1 -1
View File
@@ -7,7 +7,7 @@ ENV PYTHONUNBUFFERED=1
RUN set -eux; \
apt-get update; \
apt-get install -y git gosu; \
apt-get install -y gosu; \
rm -rf /var/lib/apt/lists/*; \
# verify that the binary works
gosu nobody true
+17 -4
View File
@@ -1,8 +1,9 @@
import logging
import os
import re
try:
import uvloop
import asyncio
import uvloop, asyncio
asyncio.set_event_loop_policy(uvloop.EventLoopPolicy())
except ImportError:
@@ -11,8 +12,6 @@ except ImportError:
BOT_TOKEN = os.getenv("BOT_TOKEN")
ADMIN = [int(i) for i in os.getenv("BOT_ADMIN").split(",")]
PIXIV_REFRESH_TOKEN = os.getenv("PIXIV_REFRESH_TOKEN")
WEBHOOK = os.getenv("WEBHOOK", False)
if WEBHOOK:
WEBHOOK_LISTEN = os.getenv("WEBHOOK_LISTEN", "0.0.0.0")
@@ -21,3 +20,17 @@ if WEBHOOK:
WEBHOOK_KEY = os.getenv("WEBHOOK_KEY", "cert/private.key")
WEBHOOK_CERT = os.getenv("WEBHOOK_CERT", "cert/cert.pem")
WEBHOOK_SECRET_TOKEN = os.getenv("WEBHOOK_SECRET_TOKEN")
x_url_regex = re.compile(r"^(?:https?://)(?:www\.|mobile\.|)(?:x|twitter|fixvx|vxtwitter)\.com/(.+)/status/(\d+)")
x_media_regex = re.compile(r"^(?:https?://)(pbs|video)\.twimg\.com/(.*)")
x_tco_regex = re.compile(r"(?:https?://)t\.co/.+$", re.M)
message_url_regex = re.compile(r"\[.+]", re.S)
logging.basicConfig(
level=os.getenv("LOG_LEVEL", "WARNING"),
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
)
def get_logger(name: str) -> logging.Logger:
return logging.getLogger(name)
+2 -3
View File
@@ -1,13 +1,12 @@
services:
tgxmb: image: yoursfunny/telegram-twitter-media-bot: latest
restart: always
# ports:
# - "8443:8443"
ports:
- "8443:8443"
environment:
LOCAL_USER_ID: '1000'
BOT_TOKEN: ''
BOT_ADMIN: ''
PIXIV_REFRESH_TOKEN: ''
WEBHOOK: false
WEBHOOK_LISTEN: '127.0.0.1'
WEBHOOK_PORT: 8443
+13 -25
View File
@@ -10,17 +10,13 @@ from telegram.ext import (ApplicationBuilder, CallbackQueryHandler, CommandHandl
InlineQueryHandler, MessageHandler, PicklePersistence, filters)
import common
import utils.regex as regex
from utils.logger import get_logger
from utils.net import NetClient
from utils.pixiv import ProcessPixiv
from utils.telegram import Telegram
from tweet import TelegramTweet
if TYPE_CHECKING:
from telegram import Chat, Message, Update
from telegram.ext import Application, ContextTypes
logger = get_logger(__name__)
logger = common.get_logger(__name__)
def send_action(action):
@@ -40,26 +36,22 @@ async def inline_query(update: Update, context: ContextTypes.DEFAULT_TYPE) -> No
if query == "":
return
logger.info(f"Query: {query}")
async with Telegram(query) as tweet:
if not tweet:
return
await update.inline_query.answer(tweet.inline_query_result())
async with TGTweet(query) as tweet:
result = list(tweet.inline_query_generator)
await update.inline_query.answer(result)
@send_action(ChatAction.UPLOAD_PHOTO)
async def url_media(update: Update, context: ContextTypes.DEFAULT_TYPE) -> None:
url = update.message.text
logger.info(f"Receiving url: {url}")
async with Telegram(url) as tweet:
media = tweet.message_media_result()
if not media:
await update.effective_message.reply_text("No media found or media type is not supported.")
return
async with TGTweet(url) as tweet:
media = list(tweet.pm_media_generator)
message_to_send = await update.effective_message.reply_media_group(
media,
caption=tweet.message_text,
reply_to_message_id=update.message.message_id,
) if not isinstance(media[0], tuple) else await update.effective_message.reply_animation(
) if not tweet.is_single_gif else await update.effective_message.reply_animation(
media[0][0],
caption=tweet.message_text,
reply_to_message_id=update.message.message_id,
@@ -113,7 +105,7 @@ async def edit_message(update: Update, context: ContextTypes.DEFAULT_TYPE) -> No
))
else:
update_text = html.escape(update.message.text)
match = regex.message_url.search(update_text)
match = common.message_url_regex.search(update_text)
if match:
match = match.span()
update_text = update_text[:match[0]] + message_url.format(
@@ -212,9 +204,7 @@ async def post_init(application: Application) -> None:
DESCRIPTION = "A bot to fetch tweets from Twitter."
await application.bot.set_my_description(DESCRIPTION)
await application.bot.set_my_short_description(DESCRIPTION)
NetClient.init_client()
if common.PIXIV_REFRESH_TOKEN:
await ProcessPixiv.init_client(common.PIXIV_REFRESH_TOKEN)
TGTweet.init_client()
async def post_stop(application: Application) -> None:
@@ -222,7 +212,7 @@ async def post_stop(application: Application) -> None:
async def post_shutdown(application: Application) -> None:
await NetClient.close_client()
await TGTweet.close_client()
def main():
@@ -236,7 +226,6 @@ def main():
.post_stop(post_stop)
.post_shutdown(post_shutdown)
.concurrent_updates(True)
.http_version('2')
.build()
)
@@ -244,9 +233,8 @@ def main():
user_filter.add_user_ids(common.ADMIN)
handlers = [
InlineQueryHandler(inline_query),
MessageHandler((filters.Regex(regex.x_url) | filters.Regex(regex.pixiv_url)) & filters.ChatType.PRIVATE,
url_media),
InlineQueryHandler(inline_query, common.x_url_regex),
MessageHandler(filters.Regex(common.x_url_regex) & filters.ChatType.PRIVATE, url_media),
CommandHandler("set_forward_channel", cmd_set_forward_channel),
CommandHandler("remove_forward_channel", cmd_remove_forward_channel),
CommandHandler("edit_before_forward", cmd_edit_before_forward),
+1 -2
View File
@@ -1,4 +1,3 @@
python-telegram-bot[webhooks]~=21.3
httpx[http2]~=0.27
httpx[http2]~=0.27.0
uvloop~=0.19.0; sys_platform != 'win32'
async-pixiv @ git+https://github.com/TheFunny/async-pixiv@main
+112 -44
View File
@@ -2,18 +2,17 @@ from __future__ import annotations
import html
from functools import cached_property
from typing import Generator, TYPE_CHECKING
from typing import TYPE_CHECKING
from uuid import uuid4
from httpx import AsyncClient
from telegram import (InlineQueryResultMpeg4Gif, InlineQueryResultPhoto, InlineQueryResultVideo, InputMediaPhoto,
InputMediaVideo)
from .logger import get_logger
from .net import NetClient
from .regex import x_media_url, x_tco_url, x_url
from common import get_logger, x_media_regex, x_tco_regex, x_url_regex
if TYPE_CHECKING:
from .types import TweetInfo, TypeInlineQueryResult, TypeMessageMediaResult
from typing import Generator, TypedDict
logger = get_logger(__name__)
@@ -26,6 +25,21 @@ message_raw_text = """{url}
"""
def create_client() -> AsyncClient:
return AsyncClient(http2=True)
async def close_client(_client: AsyncClient) -> None:
await _client.aclose()
async def fetch_json(_client: AsyncClient, url: str) -> dict:
logger.info(f"Fetching {url}")
response = await _client.get(url)
assert response.status_code == response.is_success, f"Failed to fetch {url}, status code {response.status_code}"
return response.json()
class TweetMedia:
__slots__ = ('_url', '_thumb', '_type', '__dict__')
@@ -35,11 +49,12 @@ class TweetMedia:
self._type: str = media_type
def __str__(self):
return f"TweetMedia(url={self.url}, thumb={self.thumb}, type={self.type})"
return f"Media[url: {self.url} thumb: {self.thumb} type: {self.type}]"
@cached_property
def _uri(self) -> str | None:
if match := x_media_url.match(self._url):
match = x_media_regex.match(self._url)
if match:
return match.group(2).removesuffix('.jpg').removesuffix('.png')
return None
@@ -120,10 +135,21 @@ class Tweet:
return self._sensitive
class ProcessTweet:
__slots__ = ('_url', '_tweet')
class TGTweet(Tweet):
class TweetInfo(TypedDict):
tweetID: str
user_name: str
user_screen_name: str
text: str
media_extended: list[dict]
possibly_sensitive: bool
def __init__(self, url: str):
class ProcessTweet:
__slots__ = ('_httpx_client', '_url', '_tweet')
def __init__(self, httpx_client: AsyncClient, url: str):
self._httpx_client: AsyncClient = httpx_client
self._url: str = url
async def __aenter__(self):
@@ -141,14 +167,14 @@ class ProcessTweet:
pass
async def _fetch_tweet(self) -> TweetInfo:
match = x_url.match(self._url)
match = x_url_regex.match(self._url)
assert match, f"Invalid URL: {self._url}"
auther_id, tweet_id = match.groups()
return await NetClient.fetch_json(vx_api_url.format(auther_id, tweet_id))
auther_id, tweet_id = match.group()
return await fetch_json(self._httpx_client, vx_api_url.format(auther_id, tweet_id))
@property
def _tweet_text(self) -> str:
match = x_tco_url.search(self._tweet['text'])
match = x_tco_regex.search(self._tweet['text'])
return self._tweet['text'][:match.start()].strip(" ") if match else self._tweet['text']
@property
@@ -164,42 +190,83 @@ class ProcessTweet:
class TelegramTweet:
__slots__ = ('_url', '_tweet', '__dict__')
def __init__(self, url: str):
self._url: str = url
self._api_param: tuple[str] = self._tweet_id
self._is_single_gif: bool = False
assert self._api_param
async def __aenter__(self):
async with ProcessTweet(self._url) as tweet:
self._tweet = tweet
self._tweet: dict = await self._fetch_tweet(self._api_param)
super().__init__(*self._init_properties)
return self
async def __aexit__(self, exc_type, exc_val, exc_tb):
pass
@cached_property
def url(self) -> str:
return self._tweet.url
@classmethod
def init_client(cls) -> None:
cls._httpx_client = create_client()
@cached_property
@classmethod
async def close_client(cls) -> None:
await close_client(cls._httpx_client)
async def _fetch_tweet(self, api_param: tuple[str]) -> dict:
return await fetch_json(self._httpx_client, vx_api_url.format(*api_param))
@property
def _tweet_id(self) -> tuple[str] | None:
match = x_url_regex.match(self._url)
if match:
return match.groups()
return None
@property
def _tweet_text(self) -> str:
match = x_tco_regex.search(self._tweet['text'])
return self._tweet['text'][:match.start()].strip(" ") if match else self._tweet['text']
@property
def _tweet_media(self) -> list[TweetMedia]:
return [
TweetMedia(
url=x['url'],
thumb=x['thumbnail_url'],
media_type=x['type']
)
for x in self._tweet['media_extended']
]
@property
def _init_properties(self) -> tuple:
id = self._tweet['tweetID']
author = self._tweet['user_name']
author_id = self._tweet['user_screen_name']
text = self._tweet_text
media = self._tweet_media
sensitive = self._tweet['possibly_sensitive']
return id, author, author_id, text, media, sensitive
@property
def is_single_gif(self) -> bool:
return self._is_single_gif
@property
def message_text(self) -> str:
tweet = self._tweet
return message_raw_text.format(
url=tweet.url,
author_url=tweet.author_url,
author=html.escape(tweet.author),
text=html.escape(tweet.text)
url=self.url,
author_url=self.author_url,
author=html.escape(self.author),
text=html.escape(self.text)
)
def inline_query_result(self) -> tuple[TypeInlineQueryResult, ...]:
return tuple(self.inline_query_generator())
def message_media_result(self) -> tuple[TypeMessageMediaResult, ...]:
return tuple(self.message_media_generator())
def inline_query_generator(self) -> Generator[TypeInlineQueryResult, None, None]:
tweet = self._tweet
for tweet_media in tweet.media:
@property
def inline_query_generator(self) -> Generator[
InlineQueryResultPhoto | InlineQueryResultVideo | InlineQueryResultMpeg4Gif, None, None
]:
for tweet_media in self.media:
logger.info(str(tweet_media))
if tweet_media.type == "image":
yield InlineQueryResultPhoto(
@@ -214,7 +281,7 @@ class TelegramTweet:
video_url=tweet_media.url,
mime_type="video/mp4",
thumbnail_url=tweet_media.thumb,
title=tweet.text,
title=self.text,
caption=self.message_text
)
elif tweet_media.type == "gif":
@@ -225,26 +292,27 @@ class TelegramTweet:
caption=self.message_text
)
def message_media_generator(self) -> Generator[TypeMessageMediaResult, None, None]:
tweet = self._tweet
for tweet_media in tweet.media:
@property
def pm_media_generator(self) -> Generator[InputMediaPhoto | InputMediaVideo | tuple[str, bool], None, None]:
for tweet_media in self.media:
logger.info(str(tweet_media))
if tweet_media.type == "image":
yield InputMediaPhoto(
media=tweet_media.url,
has_spoiler=tweet.sensitive
has_spoiler=self.sensitive
)
elif tweet_media.type == "video":
yield InputMediaVideo(
media=tweet_media.url,
has_spoiler=tweet.sensitive,
has_spoiler=self.sensitive,
thumbnail=tweet_media.thumb
)
elif tweet_media.type == "gif":
if len(tweet.media) == 1:
yield tweet_media.url, tweet.sensitive
if len(self.media) == 1:
self._is_single_gif = True
yield tweet_media.url, self.sensitive
yield InputMediaVideo(
media=tweet_media.url,
has_spoiler=tweet.sensitive,
has_spoiler=self.sensitive,
thumbnail=tweet_media.thumb
)
View File
-11
View File
@@ -1,11 +0,0 @@
import logging
import os
logging.basicConfig(
level=os.getenv("LOG_LEVEL", "WARNING"),
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
)
def get_logger(name: str) -> logging.Logger:
return logging.getLogger(name)
-37
View File
@@ -1,37 +0,0 @@
from __future__ import annotations
from httpx import AsyncClient
def create_client() -> AsyncClient:
return AsyncClient(http2=True)
async def close_client(_client: AsyncClient) -> None:
return await _client.aclose()
async def fetch_json(_client: AsyncClient, url: str) -> dict:
response = await _client.get(url)
assert response.is_success, f"Failed to fetch {url}, status code {response.status_code}"
return response.json()
class NetClient:
_httpx_client: AsyncClient
@classmethod
def init_client(cls) -> None:
cls._httpx_client = create_client()
@classmethod
async def close_client(cls) -> None:
await close_client(cls._httpx_client)
@classmethod
def get_client(cls) -> AsyncClient:
return cls._httpx_client
@classmethod
async def fetch_json(cls, url: str) -> dict:
return await fetch_json(cls._httpx_client, url)
-212
View File
@@ -1,212 +0,0 @@
from __future__ import annotations
import html
from typing import Generator, Literal, TYPE_CHECKING
from uuid import uuid4
from async_pixiv import PixivClient
from async_pixiv.error import ApiError
from telegram import InlineQueryResultPhoto, InputMediaPhoto
from utils.logger import get_logger
if TYPE_CHECKING:
from async_pixiv.model.illust import Illust
from utils.types import TypeInlineQueryResult, TypeMessageMediaResult
logger = get_logger(__name__)
message_raw_text = """<a href="{url}">{text}</a> / <a href="{author_url}">{author}</a>
{tags}
"""
class PixivMedia:
__slots__ = ('_url', '_thumb')
def __init__(self, url: str, thumb: str):
self._url: str = url
self._thumb: str = thumb
def __str__(self):
return f"PixivMedia(url={self.url}, thumb={self.thumb}, large={self.large})"
@property
def url(self) -> str:
return self._url
@property
def thumb(self) -> str:
return self._thumb
@property
def large(self) -> str:
url = self._url.replace("img-original", "img-master").removesuffix(".jpg").removesuffix(".png")
return url + "_master1200.jpg"
class Pixiv:
__slots__ = ('_illust',)
def __init__(self, illust: Illust):
self._illust: Illust = illust
@property
def url(self) -> str:
return str(self._illust.link).rstrip('/')
@property
def type(self) -> Literal["illust", "manga", "ugoira"]:
return self._illust.type.value
@property
def title(self) -> str:
return self._illust.title
@property
def author(self) -> str:
return self._illust.user.name
@property
def author_url(self) -> str:
return str(self._illust.user.link).rstrip('/')
@property
def description(self) -> str:
return self._illust.caption
@property
def tags(self) -> list[str]:
return [tag.name for tag in self._illust.tags]
@property
def is_multiple_pages(self) -> bool:
return self._illust.page_count > 1
@property
def is_nsfw(self) -> bool:
return self._illust.is_nsfw
@property
def is_ai(self) -> bool:
return self._illust.ai_type.value == 2
@property
def images(self) -> list[PixivMedia]:
if self.is_multiple_pages:
return [
PixivMedia(
url=page.image_urls.original,
thumb=page.image_urls.medium
)
for page in self._illust.meta_pages
]
else:
return [
PixivMedia( # should observe if image_url.original is always None for single page
url=self._illust.image_urls.original or self._illust.meta_single_page.original,
thumb=self._illust.image_urls.medium
)
]
class ProcessPixiv:
_client: PixivClient
__slots__ = ('_url', '_illust')
@classmethod
async def init_client(cls, token: str) -> None:
cls._client = PixivClient()
await cls._client.login_with_token(token)
cls._token = token
@classmethod
async def close_client(cls) -> None:
await cls._client.close()
@classmethod
async def refresh_token(cls):
await cls._client.login_with_token(cls._token)
def __init__(self, url: str):
self._url: str = url
async def __aenter__(self):
self._illust = await self._fetch_illust()
return Pixiv(self._illust)
async def __aexit__(self, exc_type, exc_val, exc_tb):
pass
async def _fetch_illust(self) -> Illust:
illust_id = self._parse_illust_id()
try:
return (await self._client.ILLUST.detail(illust_id)).illust
except ApiError:
await self.refresh_token()
return (await self._client.ILLUST.detail(illust_id)).illust # TODO use retry here
def _parse_illust_id(self) -> int:
return int(self._url.split("/")[-1])
class TelegramPixiv:
__slots__ = ('_url', '_pixiv')
def __init__(self, url: str):
self._url = url
async def __aenter__(self):
async with ProcessPixiv(self._url) as pixiv:
self._pixiv = pixiv
return self
async def __aexit__(self, exc_type, exc_val, exc_tb):
pass
def url(self) -> str:
return self._url
@property
def message_text(self) -> str:
pixiv = self._pixiv
return message_raw_text.format(
url=pixiv.url,
author_url=pixiv.author_url,
author=html.escape(pixiv.author),
text=html.escape(pixiv.title),
tags=html.escape(" ".join(f"#{name}" for name in pixiv.tags))
)
def inline_query_result(self) -> tuple[TypeInlineQueryResult, ...]:
return tuple(self.inline_query_generator())
def message_media_result(self) -> tuple[TypeMessageMediaResult, ...]:
return tuple(self.message_media_generator())
def inline_query_generator(self) -> Generator[TypeInlineQueryResult, None, None]:
pixiv = self._pixiv
for media in pixiv.images:
logger.info(str(media))
if pixiv.type in ("illust", "manga"):
yield InlineQueryResultPhoto(
id=str(uuid4()),
photo_url=media.large,
thumbnail_url=media.thumb,
caption=self.message_text
)
else:
yield
def message_media_generator(self) -> Generator[TypeMessageMediaResult, None, None]:
pixiv = self._pixiv
for media in pixiv.images:
logger.info(str(media))
if pixiv.type in ("illust", "manga"):
yield InputMediaPhoto(
media=media.large,
has_spoiler=pixiv.is_nsfw
)
else:
yield
-8
View File
@@ -1,8 +0,0 @@
import re
x_url = re.compile(r"^(?:https?://)?(?:www\.|mobile\.)?(?:x|twitter|fixvx|vxtwitter)\.com/(.+)/status/(\d+)")
x_media_url = re.compile(r"^(?:https?://)(pbs|video)\.twimg\.com/(.*)")
x_tco_url = re.compile(r"(?:https?://)t\.co/.+$", re.M)
message_url = re.compile(r"\[.+]", re.S)
pixiv_url = re.compile(r"^(?:https?://)?(?:www\.)?pixiv\.net/(?:en/)?artworks/(\d+)")
-24
View File
@@ -1,24 +0,0 @@
from __future__ import annotations
from common import PIXIV_REFRESH_TOKEN
from .pixiv import TelegramPixiv
from .regex import pixiv_url, x_url
from .tweet import TelegramTweet
class Telegram:
def __init__(self, url: str):
self._url = url
async def __aenter__(self):
if x_url.match(self._url):
async with TelegramTweet(self._url) as tweet:
return tweet
elif PIXIV_REFRESH_TOKEN and pixiv_url.match(self._url):
async with TelegramPixiv(self._url) as pixiv:
return pixiv
else:
return None
async def __aexit__(self, exc_type, exc_val, exc_tb):
pass
-17
View File
@@ -1,17 +0,0 @@
from typing import TypedDict
from telegram import InlineQueryResultMpeg4Gif, InlineQueryResultPhoto, InlineQueryResultVideo, InputMediaPhoto, \
InputMediaVideo
TypeInlineQueryResult = InlineQueryResultMpeg4Gif | InlineQueryResultPhoto | InlineQueryResultVideo
InputMediaAnimation = tuple[str, bool]
TypeMessageMediaResult = InputMediaPhoto | InputMediaVideo | InputMediaAnimation
class TweetInfo(TypedDict):
tweetID: str
user_name: str
user_screen_name: str
text: str
media_extended: list[dict]
possibly_sensitive: bool