1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-10-09 12:55:00 +02:00

feat(capture): route x.com status URLs to Source::Tweet

Full https://x.com/{user}/status/{id} URLs now go through the Twitter
JSON scraper (Source::Tweet) instead of yt-dlp (Source::X), consistent
with the x:tweet:ID shorthand. Non-status x.com URLs are unchanged.

tweet_id_from_archive_path extended to extract the numeric ID from full
status URLs, not just colon-separated shorthands.

x:media:ID remains the explicit opt-in for yt-dlp media-only capture.
This commit is contained in:
TheGeneralist 2026-10-04 18:30:47 +02:00
parent c68469783c
commit 8fc6754e28
Signed by: thegeneralist01
SSH key fingerprint: SHA256:pp9qddbCNmVNoSjevdvQvM5z0DHN7LTa8qBMbcMq/R4

View file

@ -563,6 +563,12 @@ fn determine_source(path: &str) -> Source {
} }
if path.starts_with("https://x.com/") { if path.starts_with("https://x.com/") {
// Status URLs (twitter.com/{user}/status/{id}) are tweets; everything
// else (home, search, profiles, etc.) goes through yt-dlp as Source::X.
let after_domain = &path["https://x.com/".len()..];
if after_domain.contains("/status/") {
return Source::Tweet;
}
return Source::X; return Source::X;
} }
@ -756,6 +762,17 @@ fn local_file_extension(path: &str) -> String {
} }
fn tweet_id_from_archive_path(path: &str) -> Option<String> { fn tweet_id_from_archive_path(path: &str) -> Option<String> {
// Full x.com status URL: https://x.com/{user}/status/{id}[/...]
if let Some(after_domain) = path.strip_prefix("https://x.com/") {
if let Some(status_idx) = after_domain.find("/status/") {
let id = after_domain[status_idx + "/status/".len()..]
.split('/')
.next()
.unwrap_or("");
return parse_tweet_id(id);
}
}
// Shorthand: tweet:ID, x:tweet:ID, etc.
path.split(':').next_back().and_then(parse_tweet_id) path.split(':').next_back().and_then(parse_tweet_id)
} }
@ -2692,10 +2709,20 @@ mod tests {
#[test] #[test]
fn test_x_sources() { fn test_x_sources() {
let x_cases = [ let x_cases = [
// Non-status x.com URLs still go through yt-dlp
TestCase { TestCase {
url: "https://x.com/some_post", url: "https://x.com/some_post",
expected: Source::X, expected: Source::X,
}, },
// Status URLs are tweets
TestCase {
url: "https://x.com/navyabijoy/status/2106754057834840266",
expected: Source::Tweet,
},
TestCase {
url: "https://x.com/i/status/1234567890",
expected: Source::Tweet,
},
TestCase { TestCase {
url: "x:1234567890", url: "x:1234567890",
expected: Source::Tweet, expected: Source::Tweet,