fix: expand twitter short links like FxEmbed linkFixer

Replace display_text_range slicing with FxEmbed-style content matching:
expand mapped t.co links to their real URLs (dropping internal
x.com/i/web/status pages), then strip every leftover t.co short link
(appended media link, unmapped links).

The old code cut by display_text_range, whose index unit differs per
endpoint (UTF-16 on the syndication endpoint, code points in the
GraphQL fallback), so slicing by either unit left a partial
"https://t." caption tail on the other path. Content matching is
unit-agnostic and also keeps user-posted/quote links at the end of
the text that the trailing cut previously dropped.
This commit is contained in:
2026-08-07 10:25:54 +08:00
parent fb43441c56
commit 4060a88031
3 changed files with 98 additions and 63 deletions
+5 -6
View File
@@ -228,7 +228,6 @@ fn to_syndication_shape(tweet: &Value) -> Option<Value> {
"screen_name": user.get("screen_name"), "screen_name": user.get("screen_name"),
}, },
"possibly_sensitive": legacy.get("possibly_sensitive"), "possibly_sensitive": legacy.get("possibly_sensitive"),
"display_text_range": legacy.get("display_text_range"),
"entities": legacy.get("entities"), "entities": legacy.get("entities"),
"mediaDetails": legacy.pointer("/extended_entities/media"), "mediaDetails": legacy.pointer("/extended_entities/media"),
})) }))
@@ -251,12 +250,12 @@ mod tests {
"legacy": { "legacy": {
"id_str": "2083868672721039569", "id_str": "2083868672721039569",
"full_text": "nsfw content https://t.co/abc123", "full_text": "nsfw content https://t.co/abc123",
"display_text_range": [0, 12],
"possibly_sensitive": true, "possibly_sensitive": true,
"entities": { "entities": {
"urls": [ // The appended media link lives in extended_entities.media,
{ "url": "https://t.co/abc123", "expanded_url": "https://example.com/x" } // not entities.urls, so it has no expansion mapping and the
] // content-based strip removes it.
"urls": []
}, },
"extended_entities": { "extended_entities": {
"media": [ "media": [
@@ -319,7 +318,7 @@ mod tests {
other => panic!("expected video, got {other:?}"), other => panic!("expected video, got {other:?}"),
} }
assert_eq!(fetched.source_url, "https://x.com/nsfw_author/status/2083868672721039569"); assert_eq!(fetched.source_url, "https://x.com/nsfw_author/status/2083868672721039569");
// display_text_range cuts the trailing t.co link. // The appended media short link (no URL-entity mapping) is stripped.
assert_eq!(fetched.title, "nsfw content"); assert_eq!(fetched.title, "nsfw content");
} }
+93 -53
View File
@@ -158,12 +158,10 @@ impl Tweet {
pub fn from_syndication_json(raw_json: &str) -> Result<Self, serde_json::Error> { pub fn from_syndication_json(raw_json: &str) -> Result<Self, serde_json::Error> {
let json: model::SyndicationTweet = serde_json::from_str(raw_json)?; let json: model::SyndicationTweet = serde_json::from_str(raw_json)?;
let id = json.id_str; let id = json.id_str;
// Strip the appended media short link first, then expand the remaining // Expand the user's t.co short links to their real destinations and
// t.co short links (the user's own URLs) to their real destinations. // strip the appended media short link, mirroring FxEmbed's linkFixer
let text = expand_links( // (no display_text_range arithmetic — see expand_links).
&strip_trailing_short_links(&json.text, json.display_text_range), let text = expand_links(&json.text, &json.entities.urls);
&json.entities.urls,
);
// `name` is the display name, `screen_name` the handle (Python's // `name` is the display name, `screen_name` the handle (Python's
// vxtwitter mapping: author = display name, author_id = handle). // vxtwitter mapping: author = display name, author_id = handle).
let author = json.user.name; let author = json.user.name;
@@ -204,44 +202,44 @@ impl Tweet {
} }
} }
/// The raw syndication `text` ends with the appended media short link /// Mirrors FxEmbed's `linkFixer` (link-fixer.ts): expand every t.co short
/// (" https://t.co/wmI8McgXul"). `display_text_range` marks the visible text; /// link that has an entity mapping to its real destination, drop internal
/// a regex strips any remaining trailing t.co link when the range is absent /// `x.com/i/web/status/…` plumbing links, then strip any remaining t.co
/// or a tweet ends in a URL short link. /// short link (the appended media link and other unmapped short links).
/// /// Pure content matching — no `display_text_range` arithmetic, so the
/// X reports these indices in Unicode **code points**, not UTF-16 units /// endpoint's inconsistent index units (UTF-16 vs code points, see the
/// (verified against GraphQL responses containing emoji: cutting an emoji /// deleted `strip_trailing_short_links`) never matter.
/// tweet by UTF-16 units silently drops the character after the emoji).
fn strip_trailing_short_links(text: &str, display_text_range: Option<[usize; 2]>) -> String {
let mut out = match display_text_range {
Some([start, end]) if start < end => {
text.chars().skip(start).take(end - start).collect()
}
_ => text.to_string(),
};
while TRAILING_TCO.is_match(&out) {
out = TRAILING_TCO.replace(&out, "").into_owned();
}
out
}
/// Trailing Twitter short link, optionally preceded by whitespace.
static TRAILING_TCO: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\s*https?://t\.co/[A-Za-z0-9]+$").unwrap()
});
/// Replaces every t.co short link that has an entity mapping with its
/// expanded URL. Short links without a mapping stay untouched.
fn expand_links(text: &str, urls: &[model::SyndicationEntityUrl]) -> String { fn expand_links(text: &str, urls: &[model::SyndicationEntityUrl]) -> String {
let mut out = text.to_string(); let mut out = text.to_string();
for entity in urls { for entity in urls {
if let Some(expanded) = &entity.expanded_url { let Some(expanded) = &entity.expanded_url else {
out = out.replace(&entity.url, expanded); continue;
} };
let replacement = if WEB_STATUS_URL.is_match(expanded) {
""
} else {
expanded
};
out = out.replace(&entity.url, replacement);
} }
out TCO_LINK.replace_all(&out, "").into_owned()
} }
/// Internal x.com page links (reply / quote plumbing) expand to
/// `x.com/i/web/status/<id>`; FxEmbed drops them — the tweet's own content
/// already carries the information.
static WEB_STATUS_URL: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"^https://(?:x\.com|twitter\.com)/i/web/status/\w+").unwrap()
});
/// A t.co short link, optionally preceded by a space. Any leftover
/// occurrence (unmapped — e.g. the appended media link) is removed,
/// mirroring FxEmbed. Real short-link codes are 10 alphanumerics; the
/// length-agnostic class keeps fixtures and hypothetical odd lengths safe.
static TCO_LINK: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r" ?https?://t\.co/[A-Za-z0-9]+").unwrap()
});
/// pbs.twimg.com serves a reduced default size without size params; `name=orig` /// pbs.twimg.com serves a reduced default size without size params; `name=orig`
/// returns the original file (fxtwitter used to hand out the original /// returns the original file (fxtwitter used to hand out the original
/// directly, the syndication API does not). Non-twimg URLs pass through /// directly, the syndication API does not). Non-twimg URLs pass through
@@ -411,13 +409,12 @@ mod tests {
#[test] #[test]
fn syndication_text_strips_trailing_media_short_link() { fn syndication_text_strips_trailing_media_short_link() {
// Real syndication shape: the media short link sits after the visible // Real syndication shape: the appended media short link sits after the
// text, and display_text_range marks where it begins. // visible text; the unmapped t.co link is stripped by content.
let raw = serde_json::json!({ let raw = serde_json::json!({
"__typename": "Tweet", "__typename": "Tweet",
"id_str": "1", "id_str": "1",
"text": "hello world https://t.co/abc123", "text": "hello world https://t.co/abc123",
"display_text_range": [0, 11],
"user": { "name": "N", "screen_name": "h" }, "user": { "name": "N", "screen_name": "h" },
"mediaDetails": [] "mediaDetails": []
}); });
@@ -427,8 +424,31 @@ mod tests {
} }
#[test] #[test]
fn syndication_text_strips_trailing_short_link_without_range() { fn syndication_text_strips_trailing_link_regardless_of_index_units() {
// No display_text_range: the regex fallback removes the trailing link. // Real tweet 2084567054481571919: the visible text is 30 code points
// but 41 UTF-16 units, and the two endpoints historically reported
// display_text_range in different units (UTF-16 on syndication, code
// points on GraphQL). The FxEmbed-style content-based strip ignores
// the range entirely, so the appended media link is removed for any
// response shape.
let text = "妄想𝑨𝒅𝒅𝒊𝒄𝒕𝒊𝒐𝒏…🩷💚❤️\n#ゼンゼロ #zzzero https://t.co/XnIi83EkEB";
let visible = "妄想𝑨𝒅𝒅𝒊𝒄𝒕𝒊𝒐𝒏…🩷💚❤️\n#ゼンゼロ #zzzero";
let raw = serde_json::json!({
"__typename": "Tweet",
"id_str": "2084567054481571919",
"text": text,
"user": { "name": "N", "screen_name": "h" },
"mediaDetails": []
});
let tweet = Tweet::from_syndication_json(&raw.to_string()).unwrap();
assert_eq!(tweet.text, visible, "left a partial link");
assert!(!tweet.caption().contains("t.co"));
}
#[test]
fn syndication_text_strips_trailing_short_link_without_entities() {
// No URL entities at all: the leftover t.co link is stripped by the
// content regex.
let raw = serde_json::json!({ let raw = serde_json::json!({
"__typename": "Tweet", "__typename": "Tweet",
"id_str": "1", "id_str": "1",
@@ -449,7 +469,6 @@ mod tests {
"__typename": "Tweet", "__typename": "Tweet",
"id_str": "1", "id_str": "1",
"text": "Test Tweet with @mentionThis $twtr https://t.co/RzmrQ6wAzD #hashtag https://t.co/9r69akA484", "text": "Test Tweet with @mentionThis $twtr https://t.co/RzmrQ6wAzD #hashtag https://t.co/9r69akA484",
"display_text_range": [0, 67],
"user": { "name": "N", "screen_name": "h" }, "user": { "name": "N", "screen_name": "h" },
"entities": { "entities": {
"urls": [{ "urls": [{
@@ -469,25 +488,47 @@ mod tests {
} }
#[test] #[test]
fn syndication_text_keeps_unmapped_short_links() { fn syndication_text_strips_unmapped_short_links() {
// No entity mapping for the embedded link: it stays as-is. Only the // FxEmbed parity: short links without an entity mapping (appended
// trailing media link is stripped. // media link, embedded unmapped links) are stripped, not kept.
let raw = serde_json::json!({ let raw = serde_json::json!({
"__typename": "Tweet", "__typename": "Tweet",
"id_str": "1", "id_str": "1",
"text": "check https://t.co/abc123 #tag https://t.co/def456", "text": "check https://t.co/abc123 #tag https://t.co/def456",
"display_text_range": [0, 30],
"user": { "name": "N", "screen_name": "h" }, "user": { "name": "N", "screen_name": "h" },
"mediaDetails": [] "mediaDetails": []
}); });
let tweet = Tweet::from_syndication_json(&raw.to_string()).unwrap(); let tweet = Tweet::from_syndication_json(&raw.to_string()).unwrap();
assert_eq!(tweet.text, "check https://t.co/abc123 #tag"); assert_eq!(tweet.text, "check #tag");
} }
#[test] #[test]
fn syndication_text_utf16_display_range_keeps_multibyte() { fn syndication_text_drops_internal_web_status_links() {
// display_text_range is in UTF-16 units; a Japanese text must not be // FxEmbed parity: a mapped link expanding to an internal
// sliced by UTF-8 bytes. // x.com/i/web/status/... page (reply/quote plumbing) is removed
// instead of being shown.
let raw = serde_json::json!({
"__typename": "Tweet",
"id_str": "1",
"text": "see https://t.co/xyz1234567 for context",
"user": { "name": "N", "screen_name": "h" },
"entities": {
"urls": [{
"url": "https://t.co/xyz1234567",
"expanded_url": "https://x.com/i/web/status/9876543210",
"display_url": "x.com/i/web/status/9876543210"
}]
},
"mediaDetails": []
});
let tweet = Tweet::from_syndication_json(&raw.to_string()).unwrap();
assert_eq!(tweet.text, "see for context");
assert!(!tweet.caption().contains("t.co"));
}
#[test]
fn syndication_text_keeps_multibyte_text() {
// Text-only tweet: no short links, the multibyte text is untouched.
let text = "コミティア落ちたので、明日は行きません。🙏ごめんなさい"; let text = "コミティア落ちたので、明日は行きません。🙏ごめんなさい";
let units: Vec<u16> = text.encode_utf16().collect(); let units: Vec<u16> = text.encode_utf16().collect();
assert_eq!(units.len(), 28); assert_eq!(units.len(), 28);
@@ -495,7 +536,6 @@ mod tests {
"__typename": "Tweet", "__typename": "Tweet",
"id_str": "1", "id_str": "1",
"text": text, "text": text,
"display_text_range": [0, 28],
"user": { "name": "N", "screen_name": "h" }, "user": { "name": "N", "screen_name": "h" },
"mediaDetails": [] "mediaDetails": []
}); });
-4
View File
@@ -9,10 +9,6 @@ pub struct SyndicationTweet {
pub user: SyndicationUser, pub user: SyndicationUser,
#[serde(default)] #[serde(default)]
pub possibly_sensitive: Option<bool>, pub possibly_sensitive: Option<bool>,
/// Visible-text span; the raw `text` field has the appended media short
/// link after it. Indices are Unicode code points (not UTF-16 units).
#[serde(default, rename = "display_text_range")]
pub display_text_range: Option<[usize; 2]>,
#[serde(default)] #[serde(default)]
pub entities: SyndicationEntities, pub entities: SyndicationEntities,
#[serde(default, rename = "mediaDetails")] #[serde(default, rename = "mediaDetails")]