From 4060a88031a347a5117a4947c0f19b8411a073d2 Mon Sep 17 00:00:00 2001 From: YoursFunny Date: Fri, 7 Aug 2026 10:25:54 +0800 Subject: [PATCH] fix: expand twitter short links like FxEmbed linkFixer Replace display_text_range slicing with FxEmbed-style content matching: expand mapped t.co links to their real URLs (dropping internal x.com/i/web/status pages), then strip every leftover t.co short link (appended media link, unmapped links). The old code cut by display_text_range, whose index unit differs per endpoint (UTF-16 on the syndication endpoint, code points in the GraphQL fallback), so slicing by either unit left a partial "https://t." caption tail on the other path. Content matching is unit-agnostic and also keeps user-posted/quote links at the end of the text that the trailing cut previously dropped. --- crates/x-media/src/site/twitter/auth.rs | 11 +- crates/x-media/src/site/twitter/interface.rs | 146 ++++++++++++------- crates/x-media/src/site/twitter/model.rs | 4 - 3 files changed, 98 insertions(+), 63 deletions(-) diff --git a/crates/x-media/src/site/twitter/auth.rs b/crates/x-media/src/site/twitter/auth.rs index 4458610..cfd1558 100644 --- a/crates/x-media/src/site/twitter/auth.rs +++ b/crates/x-media/src/site/twitter/auth.rs @@ -228,7 +228,6 @@ fn to_syndication_shape(tweet: &Value) -> Option { "screen_name": user.get("screen_name"), }, "possibly_sensitive": legacy.get("possibly_sensitive"), - "display_text_range": legacy.get("display_text_range"), "entities": legacy.get("entities"), "mediaDetails": legacy.pointer("/extended_entities/media"), })) @@ -251,12 +250,12 @@ mod tests { "legacy": { "id_str": "2083868672721039569", "full_text": "nsfw content https://t.co/abc123", - "display_text_range": [0, 12], "possibly_sensitive": true, "entities": { - "urls": [ - { "url": "https://t.co/abc123", "expanded_url": "https://example.com/x" } - ] + // The appended media link lives in extended_entities.media, + // not entities.urls, so it has no expansion mapping and the + // content-based strip removes it. + "urls": [] }, "extended_entities": { "media": [ @@ -319,7 +318,7 @@ mod tests { other => panic!("expected video, got {other:?}"), } assert_eq!(fetched.source_url, "https://x.com/nsfw_author/status/2083868672721039569"); - // display_text_range cuts the trailing t.co link. + // The appended media short link (no URL-entity mapping) is stripped. assert_eq!(fetched.title, "nsfw content"); } diff --git a/crates/x-media/src/site/twitter/interface.rs b/crates/x-media/src/site/twitter/interface.rs index 43ce94c..925f3e5 100644 --- a/crates/x-media/src/site/twitter/interface.rs +++ b/crates/x-media/src/site/twitter/interface.rs @@ -158,12 +158,10 @@ impl Tweet { pub fn from_syndication_json(raw_json: &str) -> Result { let json: model::SyndicationTweet = serde_json::from_str(raw_json)?; let id = json.id_str; - // Strip the appended media short link first, then expand the remaining - // t.co short links (the user's own URLs) to their real destinations. - let text = expand_links( - &strip_trailing_short_links(&json.text, json.display_text_range), - &json.entities.urls, - ); + // Expand the user's t.co short links to their real destinations and + // strip the appended media short link, mirroring FxEmbed's linkFixer + // (no display_text_range arithmetic — see expand_links). + let text = expand_links(&json.text, &json.entities.urls); // `name` is the display name, `screen_name` the handle (Python's // vxtwitter mapping: author = display name, author_id = handle). let author = json.user.name; @@ -204,44 +202,44 @@ impl Tweet { } } -/// The raw syndication `text` ends with the appended media short link -/// (" https://t.co/wmI8McgXul"). `display_text_range` marks the visible text; -/// a regex strips any remaining trailing t.co link when the range is absent -/// or a tweet ends in a URL short link. -/// -/// X reports these indices in Unicode **code points**, not UTF-16 units -/// (verified against GraphQL responses containing emoji: cutting an emoji -/// tweet by UTF-16 units silently drops the character after the emoji). -fn strip_trailing_short_links(text: &str, display_text_range: Option<[usize; 2]>) -> String { - let mut out = match display_text_range { - Some([start, end]) if start < end => { - text.chars().skip(start).take(end - start).collect() - } - _ => text.to_string(), - }; - while TRAILING_TCO.is_match(&out) { - out = TRAILING_TCO.replace(&out, "").into_owned(); - } - out -} - -/// Trailing Twitter short link, optionally preceded by whitespace. -static TRAILING_TCO: LazyLock = LazyLock::new(|| { - Regex::new(r"\s*https?://t\.co/[A-Za-z0-9]+$").unwrap() -}); - -/// Replaces every t.co short link that has an entity mapping with its -/// expanded URL. Short links without a mapping stay untouched. +/// Mirrors FxEmbed's `linkFixer` (link-fixer.ts): expand every t.co short +/// link that has an entity mapping to its real destination, drop internal +/// `x.com/i/web/status/…` plumbing links, then strip any remaining t.co +/// short link (the appended media link and other unmapped short links). +/// Pure content matching — no `display_text_range` arithmetic, so the +/// endpoint's inconsistent index units (UTF-16 vs code points, see the +/// deleted `strip_trailing_short_links`) never matter. fn expand_links(text: &str, urls: &[model::SyndicationEntityUrl]) -> String { let mut out = text.to_string(); for entity in urls { - if let Some(expanded) = &entity.expanded_url { - out = out.replace(&entity.url, expanded); - } + let Some(expanded) = &entity.expanded_url else { + continue; + }; + let replacement = if WEB_STATUS_URL.is_match(expanded) { + "" + } else { + expanded + }; + out = out.replace(&entity.url, replacement); } - out + TCO_LINK.replace_all(&out, "").into_owned() } +/// Internal x.com page links (reply / quote plumbing) expand to +/// `x.com/i/web/status/`; FxEmbed drops them — the tweet's own content +/// already carries the information. +static WEB_STATUS_URL: LazyLock = LazyLock::new(|| { + Regex::new(r"^https://(?:x\.com|twitter\.com)/i/web/status/\w+").unwrap() +}); + +/// A t.co short link, optionally preceded by a space. Any leftover +/// occurrence (unmapped — e.g. the appended media link) is removed, +/// mirroring FxEmbed. Real short-link codes are 10 alphanumerics; the +/// length-agnostic class keeps fixtures and hypothetical odd lengths safe. +static TCO_LINK: LazyLock = LazyLock::new(|| { + Regex::new(r" ?https?://t\.co/[A-Za-z0-9]+").unwrap() +}); + /// pbs.twimg.com serves a reduced default size without size params; `name=orig` /// returns the original file (fxtwitter used to hand out the original /// directly, the syndication API does not). Non-twimg URLs pass through @@ -411,13 +409,12 @@ mod tests { #[test] fn syndication_text_strips_trailing_media_short_link() { - // Real syndication shape: the media short link sits after the visible - // text, and display_text_range marks where it begins. + // Real syndication shape: the appended media short link sits after the + // visible text; the unmapped t.co link is stripped by content. let raw = serde_json::json!({ "__typename": "Tweet", "id_str": "1", "text": "hello world https://t.co/abc123", - "display_text_range": [0, 11], "user": { "name": "N", "screen_name": "h" }, "mediaDetails": [] }); @@ -427,8 +424,31 @@ mod tests { } #[test] - fn syndication_text_strips_trailing_short_link_without_range() { - // No display_text_range: the regex fallback removes the trailing link. + fn syndication_text_strips_trailing_link_regardless_of_index_units() { + // Real tweet 2084567054481571919: the visible text is 30 code points + // but 41 UTF-16 units, and the two endpoints historically reported + // display_text_range in different units (UTF-16 on syndication, code + // points on GraphQL). The FxEmbed-style content-based strip ignores + // the range entirely, so the appended media link is removed for any + // response shape. + let text = "妄想𝑨𝒅𝒅𝒊𝒄𝒕𝒊𝒐𝒏…🩷💚❤️\n#ゼンゼロ #zzzero https://t.co/XnIi83EkEB"; + let visible = "妄想𝑨𝒅𝒅𝒊𝒄𝒕𝒊𝒐𝒏…🩷💚❤️\n#ゼンゼロ #zzzero"; + let raw = serde_json::json!({ + "__typename": "Tweet", + "id_str": "2084567054481571919", + "text": text, + "user": { "name": "N", "screen_name": "h" }, + "mediaDetails": [] + }); + let tweet = Tweet::from_syndication_json(&raw.to_string()).unwrap(); + assert_eq!(tweet.text, visible, "left a partial link"); + assert!(!tweet.caption().contains("t.co")); + } + + #[test] + fn syndication_text_strips_trailing_short_link_without_entities() { + // No URL entities at all: the leftover t.co link is stripped by the + // content regex. let raw = serde_json::json!({ "__typename": "Tweet", "id_str": "1", @@ -449,7 +469,6 @@ mod tests { "__typename": "Tweet", "id_str": "1", "text": "Test Tweet with @mentionThis $twtr https://t.co/RzmrQ6wAzD #hashtag https://t.co/9r69akA484", - "display_text_range": [0, 67], "user": { "name": "N", "screen_name": "h" }, "entities": { "urls": [{ @@ -469,25 +488,47 @@ mod tests { } #[test] - fn syndication_text_keeps_unmapped_short_links() { - // No entity mapping for the embedded link: it stays as-is. Only the - // trailing media link is stripped. + fn syndication_text_strips_unmapped_short_links() { + // FxEmbed parity: short links without an entity mapping (appended + // media link, embedded unmapped links) are stripped, not kept. let raw = serde_json::json!({ "__typename": "Tweet", "id_str": "1", "text": "check https://t.co/abc123 #tag https://t.co/def456", - "display_text_range": [0, 30], "user": { "name": "N", "screen_name": "h" }, "mediaDetails": [] }); let tweet = Tweet::from_syndication_json(&raw.to_string()).unwrap(); - assert_eq!(tweet.text, "check https://t.co/abc123 #tag"); + assert_eq!(tweet.text, "check #tag"); } #[test] - fn syndication_text_utf16_display_range_keeps_multibyte() { - // display_text_range is in UTF-16 units; a Japanese text must not be - // sliced by UTF-8 bytes. + fn syndication_text_drops_internal_web_status_links() { + // FxEmbed parity: a mapped link expanding to an internal + // x.com/i/web/status/... page (reply/quote plumbing) is removed + // instead of being shown. + let raw = serde_json::json!({ + "__typename": "Tweet", + "id_str": "1", + "text": "see https://t.co/xyz1234567 for context", + "user": { "name": "N", "screen_name": "h" }, + "entities": { + "urls": [{ + "url": "https://t.co/xyz1234567", + "expanded_url": "https://x.com/i/web/status/9876543210", + "display_url": "x.com/i/web/status/9876543210" + }] + }, + "mediaDetails": [] + }); + let tweet = Tweet::from_syndication_json(&raw.to_string()).unwrap(); + assert_eq!(tweet.text, "see for context"); + assert!(!tweet.caption().contains("t.co")); + } + + #[test] + fn syndication_text_keeps_multibyte_text() { + // Text-only tweet: no short links, the multibyte text is untouched. let text = "コミティア落ちたので、明日は行きません。🙏ごめんなさい"; let units: Vec = text.encode_utf16().collect(); assert_eq!(units.len(), 28); @@ -495,7 +536,6 @@ mod tests { "__typename": "Tweet", "id_str": "1", "text": text, - "display_text_range": [0, 28], "user": { "name": "N", "screen_name": "h" }, "mediaDetails": [] }); diff --git a/crates/x-media/src/site/twitter/model.rs b/crates/x-media/src/site/twitter/model.rs index bef9d60..1d62858 100644 --- a/crates/x-media/src/site/twitter/model.rs +++ b/crates/x-media/src/site/twitter/model.rs @@ -9,10 +9,6 @@ pub struct SyndicationTweet { pub user: SyndicationUser, #[serde(default)] pub possibly_sensitive: Option, - /// Visible-text span; the raw `text` field has the appended media short - /// link after it. Indices are Unicode code points (not UTF-16 units). - #[serde(default, rename = "display_text_range")] - pub display_text_range: Option<[usize; 2]>, #[serde(default)] pub entities: SyndicationEntities, #[serde(default, rename = "mediaDetails")]