1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-10-09 12:55:00 +02:00

feat(core): add more Freedium domains + restrict Freedium routing to supported domains (#36)

Previously, any WebPage capture with via_freedium=true was routed
through the Freedium mirror regardless of the URL's host. This caused
non-paywall sites like borretti.me to be fetched through Freedium,
which is wrong — Freedium only knows how to handle a specific set of
publications.

Add is_freedium_supported_url() with a static allowlist of the 7 hosts
Freedium explicitly supports per its homepage announcement:

  Medium, NYT, WaPo, Bloomberg, Reuters, Economist, Financial Times

The gate matches on the bare domain or any subdomain (e.g.
towardsdatascience.medium.com). Unparseable URLs fall through to a
direct fetch. The existing freedium-mirror.cfd re-wrap guard is kept
as a belt-and-suspenders check after the new allowlist test.

Fixes: https://borretti.me/article/notes-on-managing-adhd archived via
Freedium despite not being a Medium article.
This commit is contained in:
TheGeneralist 2026-07-29 17:40:07 +02:00 • committed by GitHub
parent 6a0f59ea94
commit 95cd46978d
No known key found for this signature in database
GPG key ID: B5690EEEBB952194

View file

@ -1,13 +1,17 @@
use crate::{
archive::{self, ArchivePaths},
database, downloader,
twitter::parse_tweet_id,
};
use anyhow::{Context, Result}; use anyhow::{Context, Result};
use chrono::Local; use chrono::Local;
use uuid::Uuid;
use serde_json::json; use serde_json::json;
use std::{ use std::{
collections::{HashMap, HashSet}, collections::{HashMap, HashSet},
fs, fs,
path::{Path, PathBuf}, path::{Path, PathBuf},
}; };
use crate::{archive::{self, ArchivePaths}, database, downloader, twitter::parse_tweet_id}; use uuid::Uuid;
#[derive(Debug, PartialEq, Eq, Clone, Copy)] #[derive(Debug, PartialEq, Eq, Clone, Copy)]
pub enum Source { pub enum Source {
@ -125,15 +129,12 @@ pub fn resolve_cookies_for_url(
None => true, None => true,
Some(pattern) => match rule.pattern_kind.as_str() { Some(pattern) => match rule.pattern_kind.as_str() {
"wildcard" => wildcard_matches(pattern, url), "wildcard" => wildcard_matches(pattern, url),
"regex" => regex::Regex::new(pattern) "regex" => regex::Regex::new(pattern).is_ok_and(|re| re.is_match(url)),
.is_ok_and(|re| re.is_match(url)),
_ => false, _ => false,
}, },
}; };
if applies { if applies {
if let Ok(map) = if let Ok(map) = serde_json::from_str::<HashMap<String, String>>(&rule.cookies_json) {
serde_json::from_str::<HashMap<String, String>>(&rule.cookies_json)
{
result.extend(map); result.extend(map);
} }
} }
@ -163,7 +164,10 @@ fn wildcard_matches(pattern: &str, url: &str) -> bool {
match ch { match ch {
'*' => pat.push_str(".*"), '*' => pat.push_str(".*"),
'?' => pat.push('.'), '?' => pat.push('.'),
c if "$.+[]{}()|^\\".contains(c) => { pat.push('\\'); pat.push(c); } c if "$.+[]{}()|^\\".contains(c) => {
pat.push('\\');
pat.push(c);
}
c => pat.push(c), c => pat.push(c),
} }
} }
@ -171,10 +175,48 @@ fn wildcard_matches(pattern: &str, url: &str) -> bool {
regex::Regex::new(&pat).is_ok_and(|re| re.is_match(&target)) regex::Regex::new(&pat).is_ok_and(|re| re.is_match(&target))
} }
/// The set of base domains that Freedium supports as of its "7 new sources" announcement.
/// A URL matches if its hostname is exactly `domain` or any subdomain (`*.domain`).
///
/// Source: https://freedium-mirror.cfd/ — "7 new sources supported! Medium, NYT, WaPo,
/// Bloomberg, Reuters, Economist, and Financial Times."
const FREEDIUM_SUPPORTED_HOSTS: &[&str] = &[
"medium.com",
"nytimes.com",
"washingtonpost.com",
"bloomberg.com",
"reuters.com",
"economist.com",
"ft.com",
];
/// Returns true when `url`'s hostname is one of the domains Freedium explicitly supports,
/// either as the bare domain or a subdomain (e.g. `towardsdatascience.medium.com`).
/// Unparseable URLs return false so we never accidentally proxy an unknown host.
fn is_freedium_supported_url(url: &str) -> bool {
use reqwest::Url as ReqwestUrl;
let host = match ReqwestUrl::parse(url)
.ok()
.and_then(|u| u.host_str().map(str::to_string))
{
Some(h) => h,
None => return false,
};
FREEDIUM_SUPPORTED_HOSTS
.iter()
.any(|&domain| host == domain || host.ends_with(&format!(".{domain}")))
}
fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String { fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
match source { match source {
Source::YouTubeVideo => meta.title.clone().unwrap_or_else(|| "YouTube Video".to_string()), Source::YouTubeVideo => meta
Source::YouTubePlaylist => meta.title.clone().unwrap_or_else(|| "YouTube Playlist".to_string()), .title
.clone()
.unwrap_or_else(|| "YouTube Video".to_string()),
Source::YouTubePlaylist => meta
.title
.clone()
.unwrap_or_else(|| "YouTube Playlist".to_string()),
Source::YouTubeChannel => format!( Source::YouTubeChannel => format!(
"Archival of {}", "Archival of {}",
meta.author.as_deref().unwrap_or("Unknown Channel") meta.author.as_deref().unwrap_or("Unknown Channel")
@ -186,18 +228,28 @@ fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
None => title.to_string(), None => title.to_string(),
} }
} }
Source::YouTubeMusicPlaylist => { Source::YouTubeMusicPlaylist => meta
meta.title.clone().unwrap_or_else(|| "YouTube Music Playlist".to_string()) .title
} .clone()
Source::SpotifyTrack | Source::SpotifyAlbum | Source::SpotifyPlaylist => { .unwrap_or_else(|| "YouTube Music Playlist".to_string()),
meta.title.clone().unwrap_or_else(|| "Spotify Content".to_string()) Source::SpotifyTrack | Source::SpotifyAlbum | Source::SpotifyPlaylist => meta
} .title
.clone()
.unwrap_or_else(|| "Spotify Content".to_string()),
Source::X => format!("X Media by {}", meta.author.as_deref().unwrap_or("unknown")), Source::X => format!("X Media by {}", meta.author.as_deref().unwrap_or("unknown")),
Source::Tweet => { Source::Tweet => {
let excerpt = meta.caption_excerpt().unwrap_or_else(|| "Tweet".to_string()); let excerpt = meta
format!("{} \u{2014} @{}", excerpt, meta.author.as_deref().unwrap_or("unknown")) .caption_excerpt()
.unwrap_or_else(|| "Tweet".to_string());
format!(
"{} \u{2014} @{}",
excerpt,
meta.author.as_deref().unwrap_or("unknown")
)
}
Source::TweetThread => {
format!("Thread by @{}", meta.author.as_deref().unwrap_or("unknown"))
} }
Source::TweetThread => format!("Thread by @{}", meta.author.as_deref().unwrap_or("unknown")),
Source::Instagram => format!("Post by @{}", meta.author.as_deref().unwrap_or("unknown")), Source::Instagram => format!("Post by @{}", meta.author.as_deref().unwrap_or("unknown")),
Source::Facebook => format!("Post by {}", meta.author.as_deref().unwrap_or("unknown")), Source::Facebook => format!("Post by {}", meta.author.as_deref().unwrap_or("unknown")),
Source::TikTok => format!("TikTok by @{}", meta.author.as_deref().unwrap_or("unknown")), Source::TikTok => format!("TikTok by @{}", meta.author.as_deref().unwrap_or("unknown")),
@ -208,23 +260,39 @@ fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
meta.post_author.as_deref().unwrap_or("unknown") meta.post_author.as_deref().unwrap_or("unknown")
), ),
Source::Snapchat => format!("Snap by {}", meta.author.as_deref().unwrap_or("unknown")), Source::Snapchat => format!("Snap by {}", meta.author.as_deref().unwrap_or("unknown")),
Source::Local => meta.title.clone().unwrap_or_else(|| "Local File".to_string()), Source::Local => meta
Source::Url => meta.title.clone().unwrap_or_else(|| "Downloaded File".to_string()), .title
Source::WebPage => meta.title.clone().unwrap_or_else(|| "Archived Web Page".to_string()), .clone()
.unwrap_or_else(|| "Local File".to_string()),
Source::Url => meta
.title
.clone()
.unwrap_or_else(|| "Downloaded File".to_string()),
Source::WebPage => meta
.title
.clone()
.unwrap_or_else(|| "Archived Web Page".to_string()),
Source::Other => "Archived Content".to_string(), Source::Other => "Archived Content".to_string(),
} }
} }
/// Returns true when `s` is a valid bare YouTube video ID: /// Returns true when `s` is a valid bare YouTube video ID:
/// exactly 11 characters from the set `[A-Za-z0-9_-]`. /// exactly 11 characters from the set `[A-Za-z0-9_-]`.
fn is_youtube_video_id(s: &str) -> bool { fn is_youtube_video_id(s: &str) -> bool {
s.len() == 11 && s.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'_' || b == b'-') s.len() == 11
&& s.bytes()
.all(|b| b.is_ascii_alphanumeric() || b == b'_' || b == b'-')
} }
fn expand_shorthand_to_url(path: &str, source: &Source) -> String { fn expand_shorthand_to_url(path: &str, source: &Source) -> String {
// YouTube shorthands: yt:video/ID, yt:playlist/ID, yt:@handle, yt:channel/ID, etc. // YouTube shorthands: yt:video/ID, yt:playlist/ID, yt:@handle, yt:channel/ID, etc.
if matches!(source, Source::YouTubeVideo | Source::YouTubePlaylist | Source::YouTubeChannel) { if matches!(
if let Some(after) = path.strip_prefix("yt:").or_else(|| path.strip_prefix("youtube:")) { source,
Source::YouTubeVideo | Source::YouTubePlaylist | Source::YouTubeChannel
) {
if let Some(after) = path
.strip_prefix("yt:")
.or_else(|| path.strip_prefix("youtube:"))
{
if let Some(id) = after if let Some(id) = after
.strip_prefix("video/") .strip_prefix("video/")
.or_else(|| after.strip_prefix("short/")) .or_else(|| after.strip_prefix("short/"))
@ -255,7 +323,10 @@ fn expand_shorthand_to_url(path: &str, source: &Source) -> String {
} }
// YouTube Music shorthands: ytm:ID (track) or ytm:playlist/ID // YouTube Music shorthands: ytm:ID (track) or ytm:playlist/ID
if matches!(source, Source::YouTubeMusicTrack | Source::YouTubeMusicPlaylist) { if matches!(
source,
Source::YouTubeMusicTrack | Source::YouTubeMusicPlaylist
) {
if let Some(after) = path.strip_prefix("ytm:") { if let Some(after) = path.strip_prefix("ytm:") {
if let Some(id) = after.strip_prefix("playlist/") { if let Some(id) = after.strip_prefix("playlist/") {
return format!("https://music.youtube.com/playlist?list={id}"); return format!("https://music.youtube.com/playlist?list={id}");
@ -266,7 +337,10 @@ fn expand_shorthand_to_url(path: &str, source: &Source) -> String {
} }
// Spotify shorthands: spotify:track:ID, spotify:album:ID, spotify:playlist:ID // Spotify shorthands: spotify:track:ID, spotify:album:ID, spotify:playlist:ID
if matches!(source, Source::SpotifyTrack | Source::SpotifyAlbum | Source::SpotifyPlaylist) { if matches!(
source,
Source::SpotifyTrack | Source::SpotifyAlbum | Source::SpotifyPlaylist
) {
if let Some(after) = path.strip_prefix("spotify:") { if let Some(after) = path.strip_prefix("spotify:") {
if let Some(id) = after.strip_prefix("track:") { if let Some(id) = after.strip_prefix("track:") {
return format!("https://open.spotify.com/track/{id}"); return format!("https://open.spotify.com/track/{id}");
@ -869,8 +943,12 @@ fn register_tweet_artifacts(
}, },
)?; )?;
let json_path = store_path.join(relpath); let json_path = store_path.join(relpath);
let json_str = fs::read_to_string(&json_path) let json_str = fs::read_to_string(&json_path).with_context(|| {
.with_context(|| format!("failed to read tweet JSON for artifact registration: {}", json_path.display()))?; format!(
"failed to read tweet JSON for artifact registration: {}",
json_path.display()
)
})?;
for (role, raw_relpath) in tweet_raw_artifacts(&json_str)? { for (role, raw_relpath) in tweet_raw_artifacts(&json_str)? {
let raw_path = PathBuf::from(&raw_relpath); let raw_path = PathBuf::from(&raw_relpath);
let blob = blob_record_for_raw_relpath(store_path, &raw_path)?; let blob = blob_record_for_raw_relpath(store_path, &raw_path)?;
@ -1081,10 +1159,12 @@ pub fn perform_capture(
} }
}; };
let container_title = playlist_info let container_title = playlist_info.title.clone().or_else(|| {
.title playlist_info
.uploader
.clone() .clone()
.or_else(|| playlist_info.uploader.clone().map(|u| format!("{u} (channel)"))); .map(|u| format!("{u} (channel)"))
});
// Sync mode: reuse an existing container so we don't create a duplicate // Sync mode: reuse an existing container so we don't create a duplicate
// root entry on every sync run. Non-sync always creates a fresh container. // root entry on every sync run. Non-sync always creates a fresh container.
@ -1092,55 +1172,89 @@ pub fn perform_capture(
match archive::find_container_entry_id_by_canonical_url(&conn, &canonical_url) { match archive::find_container_entry_id_by_canonical_url(&conn, &canonical_url) {
Err(e) => { Err(e) => {
return Err(fail_run( return Err(fail_run(
&conn, &run, &item, &conn,
&run,
&item,
&format!("Failed to query existing container: {e:#}"), &format!("Failed to query existing container: {e:#}"),
)); ));
} }
Ok(Some(existing_id)) => { Ok(Some(existing_id)) => {
// Container already exists — mark the run item done pointing at it; // Container already exists — mark the run item done pointing at it;
// don't call record_container_entry (that would create a duplicate). // don't call record_container_entry (that would create a duplicate).
if let Err(e) = database::complete_archive_run_item(&conn, item.id, existing_id) { if let Err(e) = database::complete_archive_run_item(&conn, item.id, existing_id)
{
return Err(fail_run( return Err(fail_run(
&conn, &run, &item, &conn,
&run,
&item,
&format!("Failed to complete run item for existing container: {e:#}"), &format!("Failed to complete run item for existing container: {e:#}"),
)); ));
} }
let archived = match archive::get_archived_playlist_child_urls(&conn, &canonical_url) { let archived =
match archive::get_archived_playlist_child_urls(&conn, &canonical_url) {
Ok(set) => set, Ok(set) => set,
Err(e) => return Err(fail_run( Err(e) => {
&conn, &run, &item, return Err(fail_run(
&conn,
&run,
&item,
&format!("Failed to query archived playlist children: {e:#}"), &format!("Failed to query archived playlist children: {e:#}"),
)), ));
}
}; };
(existing_id, archived) (existing_id, archived)
} }
Ok(None) => { Ok(None) => {
// First sync run for this playlist — create the container normally. // First sync run for this playlist — create the container normally.
let e = match record_container_entry( let e = match record_container_entry(
&conn, store_path, user_id, &run, &item, locator, &canonical_url, &conn,
source, container_title, &playlist_info.playlist_id, store_path,
user_id,
&run,
&item,
locator,
&canonical_url,
source,
container_title,
&playlist_info.playlist_id,
playlist_info.uploader.as_deref(), playlist_info.uploader.as_deref(),
) { ) {
Ok(e) => e, Ok(e) => e,
Err(e) => return Err(fail_run( Err(e) => {
&conn, &run, &item, return Err(fail_run(
&conn,
&run,
&item,
&format!("Failed to create container entry: {e:#}"), &format!("Failed to create container entry: {e:#}"),
)), ));
}
}; };
(e.id, std::collections::HashSet::new()) (e.id, std::collections::HashSet::new())
} }
} }
} else { } else {
let e = match record_container_entry( let e = match record_container_entry(
&conn, store_path, user_id, &run, &item, locator, &canonical_url, &conn,
source, container_title, &playlist_info.playlist_id, store_path,
user_id,
&run,
&item,
locator,
&canonical_url,
source,
container_title,
&playlist_info.playlist_id,
playlist_info.uploader.as_deref(), playlist_info.uploader.as_deref(),
) { ) {
Ok(e) => e, Ok(e) => e,
Err(e) => return Err(fail_run( Err(e) => {
&conn, &run, &item, return Err(fail_run(
&conn,
&run,
&item,
&format!("Failed to create container entry: {e:#}"), &format!("Failed to create container entry: {e:#}"),
)), ));
}
}; };
(e.id, std::collections::HashSet::new()) (e.id, std::collections::HashSet::new())
}; };
@ -1168,7 +1282,6 @@ pub fn perform_capture(
continue; continue;
} }
let child_timestamp = format!( let child_timestamp = format!(
"{}-{}", "{}-{}",
Local::now().format("%Y-%m-%dT%H-%M-%S%.3f"), Local::now().format("%Y-%m-%dT%H-%M-%S%.3f"),
@ -1187,7 +1300,10 @@ pub fn perform_capture(
) { ) {
Ok(i) => i, Ok(i) => i,
Err(e) => { Err(e) => {
eprintln!("warn: playlist item {} create_run_item failed: {e:#}", playlist_item.url); eprintln!(
"warn: playlist item {} create_run_item failed: {e:#}",
playlist_item.url
);
continue; continue;
} }
}; };
@ -1241,7 +1357,8 @@ pub fn perform_capture(
Err(e) => { Err(e) => {
eprintln!("warn: stat child temp file: {e:#}"); eprintln!("warn: stat child temp file: {e:#}");
let _ = database::fail_archive_run_item( let _ = database::fail_archive_run_item(
&conn, child_item.id, &conn,
child_item.id,
&format!("failed to stat downloaded file: {e:#}"), &format!("failed to stat downloaded file: {e:#}"),
); );
continue; continue;
@ -1251,7 +1368,8 @@ pub fn perform_capture(
if let Err(e) = move_temp_to_raw(&temp_file, &hash, store_path) { if let Err(e) = move_temp_to_raw(&temp_file, &hash, store_path) {
eprintln!("warn: move_temp_to_raw child: {e:#}"); eprintln!("warn: move_temp_to_raw child: {e:#}");
let _ = database::fail_archive_run_item( let _ = database::fail_archive_run_item(
&conn, child_item.id, &conn,
child_item.id,
&format!("failed to move downloaded file: {e:#}"), &format!("failed to move downloaded file: {e:#}"),
); );
continue; continue;
@ -1286,7 +1404,8 @@ pub fn perform_capture(
Err(e) => { Err(e) => {
eprintln!("warn: record child entry: {e:#}"); eprintln!("warn: record child entry: {e:#}");
let _ = database::fail_archive_run_item( let _ = database::fail_archive_run_item(
&conn, child_item.id, &conn,
child_item.id,
&format!("failed to record entry: {e:#}"), &format!("failed to record entry: {e:#}"),
); );
} }
@ -1294,9 +1413,13 @@ pub fn perform_capture(
} }
Err(e) => { Err(e) => {
let _ = fs::remove_dir_all(store_path.join("temp").join(&child_timestamp)); let _ = fs::remove_dir_all(store_path.join("temp").join(&child_timestamp));
eprintln!("warn: yt-dlp child download failed for {}: {e:#}", playlist_item.url); eprintln!(
"warn: yt-dlp child download failed for {}: {e:#}",
playlist_item.url
);
let _ = database::fail_archive_run_item( let _ = database::fail_archive_run_item(
&conn, child_item.id, &conn,
child_item.id,
&format!("yt-dlp download failed: {e:#}"), &format!("yt-dlp download failed: {e:#}"),
); );
} }
@ -1313,8 +1436,8 @@ pub fn perform_capture(
|row| row.get(0), |row| row.get(0),
) )
.unwrap_or_else(|_| "completed".to_string()); .unwrap_or_else(|_| "completed".to_string());
let completed_child_count: i64 = database::get_run_completed_child_count(&conn, run.id) let completed_child_count: i64 =
.unwrap_or(0); database::get_run_completed_child_count(&conn, run.id).unwrap_or(0);
return Ok(CaptureResult { return Ok(CaptureResult {
run_uid: run.run_uid.clone(), run_uid: run.run_uid.clone(),
@ -1391,20 +1514,36 @@ pub fn perform_capture(
// Source: web page — archive as a self-contained HTML snapshot via single-file-cli // Source: web page — archive as a self-contained HTML snapshot via single-file-cli
if source == Source::WebPage { if source == Source::WebPage {
// When via_freedium is enabled and the URL is not already a freedium mirror, // When via_freedium is enabled, the URL is on a Freedium-supported host, and it is
// fetch through the mirror to bypass paywalls. Store the original locator in DB. // not already a freedium mirror URL, fetch through the mirror to bypass paywalls.
// Store the original locator in the DB; only the fetch URL changes.
// Use an empty cookie jar for the mirror URL: cookies resolved for the original // Use an empty cookie jar for the mirror URL: cookies resolved for the original
// domain (e.g. NYT, Medium) must not be sent to freedium-mirror.cfd. // domain (e.g. NYT, Medium) must not be sent to freedium-mirror.cfd.
let (fetch_url, fetch_cookies): (String, HashMap<String, String>) = let (fetch_url, fetch_cookies): (String, HashMap<String, String>) = if config.via_freedium
if config.via_freedium && !locator.starts_with("https://freedium-mirror.cfd/") { && is_freedium_supported_url(locator)
(format!("https://freedium-mirror.cfd/{}", locator), HashMap::new()) && !locator.starts_with("https://freedium-mirror.cfd/")
{
(
format!("https://freedium-mirror.cfd/{}", locator),
HashMap::new(),
)
} else { } else {
(locator.to_string(), cookies.clone()) (locator.to_string(), cookies.clone())
}; };
// Key cleanup/title-stripping off the actual fetch host so that // Key cleanup/title-stripping off the actual fetch host so that
// user-supplied freedium-mirror.cfd URLs are also handled correctly. // user-supplied freedium-mirror.cfd URLs are also handled correctly.
let is_freedium_fetch = fetch_url.starts_with("https://freedium-mirror.cfd/"); let is_freedium_fetch = fetch_url.starts_with("https://freedium-mirror.cfd/");
match downloader::singlefile::save(&fetch_url, store_path, &timestamp, &fetch_cookies, config.ublock_enabled, config.cookie_ext_enabled, config.reader_mode, config.modal_closer_enabled, is_freedium_fetch) { match downloader::singlefile::save(
&fetch_url,
store_path,
&timestamp,
&fetch_cookies,
config.ublock_enabled,
config.cookie_ext_enabled,
config.reader_mode,
config.modal_closer_enabled,
is_freedium_fetch,
) {
Ok(result) => { Ok(result) => {
let file_extension = ".html".to_string(); let file_extension = ".html".to_string();
let temp_html = store_path let temp_html = store_path
@ -1415,8 +1554,9 @@ pub fn perform_capture(
// Font extraction: rewrite the HTML in-place before hashing. // Font extraction: rewrite the HTML in-place before hashing.
// Only runs when archive_id is known (server context). CLI passes // Only runs when archive_id is known (server context). CLI passes
// None and keeps fonts embedded — no behaviour change for CLI. // None and keeps fonts embedded — no behaviour change for CLI.
let (html_hash, byte_size, extracted_fonts, html_title) = let (html_hash, byte_size, extracted_fonts, html_title) = if let Some(aid) =
if let Some(aid) = archive_id { archive_id
{
let content = fs::read_to_string(&temp_html) let content = fs::read_to_string(&temp_html)
.with_context(|| format!("failed to read {}", temp_html.display()))?; .with_context(|| format!("failed to read {}", temp_html.display()))?;
let (rewritten, fonts) = let (rewritten, fonts) =
@ -1448,9 +1588,12 @@ pub fn perform_capture(
let fav_hash = result.favicon_hash.as_deref()?; let fav_hash = result.favicon_hash.as_deref()?;
let fav_ext = result.favicon_ext.as_deref()?; let fav_ext = result.favicon_ext.as_deref()?;
let fav_temp = store_path let fav_temp = store_path
.join("temp").join(&timestamp) .join("temp")
.join(&timestamp)
.join(format!("{timestamp}.favicon{fav_ext}")); .join(format!("{timestamp}.favicon{fav_ext}"));
if !fav_temp.exists() { return None; } if !fav_temp.exists() {
return None;
}
let fav_size = fs::metadata(&fav_temp).ok()?.len() as i64; let fav_size = fs::metadata(&fav_temp).ok()?.len() as i64;
let fav_raw = raw_relative_path_from_hash(fav_hash, fav_ext).ok()?; let fav_raw = raw_relative_path_from_hash(fav_hash, fav_ext).ok()?;
if !hash_exists(fav_hash, fav_ext, store_path).ok()? { if !hash_exists(fav_hash, fav_ext, store_path).ok()? {
@ -1658,7 +1801,13 @@ pub fn perform_capture(
| Source::TikTok | Source::TikTok
| Source::Reddit | Source::Reddit
| Source::Snapchat => { | Source::Snapchat => {
match downloader::ytdlp::download(path.clone(), store_path, &timestamp, quality, &cookies) { match downloader::ytdlp::download(
path.clone(),
store_path,
&timestamp,
quality,
&cookies,
) {
Ok(result) => result, Ok(result) => result,
Err(e) => { Err(e) => {
return Err(fail_run( return Err(fail_run(
@ -1672,7 +1821,13 @@ pub fn perform_capture(
} }
Source::YouTubeMusicTrack | Source::SpotifyTrack => { Source::YouTubeMusicTrack | Source::SpotifyTrack => {
// Music tracks are always audio-only regardless of the caller's quality hint. // Music tracks are always audio-only regardless of the caller's quality hint.
match downloader::ytdlp::download(path.clone(), store_path, &timestamp, Some("audio"), &cookies) { match downloader::ytdlp::download(
path.clone(),
store_path,
&timestamp,
Some("audio"),
&cookies,
) {
Ok(result) => result, Ok(result) => result,
Err(e) => { Err(e) => {
return Err(fail_run( return Err(fail_run(
@ -1684,8 +1839,7 @@ pub fn perform_capture(
} }
} }
} }
Source::Local => { Source::Local => match downloader::local::save(path.clone(), store_path, &timestamp) {
match downloader::local::save(path.clone(), store_path, &timestamp) {
Ok(h) => (h, local_file_extension(&path)), Ok(h) => (h, local_file_extension(&path)),
Err(e) => { Err(e) => {
return Err(fail_run( return Err(fail_run(
@ -1695,8 +1849,7 @@ pub fn perform_capture(
&format!("Failed to archive local file: {e}"), &format!("Failed to archive local file: {e}"),
)); ));
} }
} },
}
Source::YouTubePlaylist | Source::YouTubeChannel => unreachable!(), Source::YouTubePlaylist | Source::YouTubeChannel => unreachable!(),
_ => unreachable!(), _ => unreachable!(),
}; };
@ -1811,10 +1964,7 @@ pub fn perform_rearchive(
// Parse source_metadata_json for tweet_id and requested_locator. // Parse source_metadata_json for tweet_id and requested_locator.
let meta: serde_json::Value = serde_json::from_str(&entry.source_metadata_json) let meta: serde_json::Value = serde_json::from_str(&entry.source_metadata_json)
.unwrap_or(serde_json::Value::Object(Default::default())); .unwrap_or(serde_json::Value::Object(Default::default()));
let requested_locator = meta["requested_locator"] let requested_locator = meta["requested_locator"].as_str().unwrap_or("").to_string();
.as_str()
.unwrap_or("")
.to_string();
let tweet_id = meta["tweet_id"].as_str().unwrap_or("").to_string(); let tweet_id = meta["tweet_id"].as_str().unwrap_or("").to_string();
if requested_locator.is_empty() || tweet_id.is_empty() { if requested_locator.is_empty() || tweet_id.is_empty() {
return Ok(RearchiveResult { return Ok(RearchiveResult {
@ -1858,7 +2008,10 @@ pub fn perform_rearchive(
database::refresh_entry_cached_bytes(&conn, entry.id)?; database::refresh_entry_cached_bytes(&conn, entry.id)?;
eprintln!("info: rearchived entry {entry_uid}: {} tweet JSONs", tweet_json_relpaths.len()); eprintln!(
"info: rearchived entry {entry_uid}: {} tweet JSONs",
tweet_json_relpaths.len()
);
Ok(RearchiveResult { Ok(RearchiveResult {
status: "completed".to_string(), status: "completed".to_string(),
@ -2012,7 +2165,10 @@ mod tests {
); );
// Full YouTube URLs pass through unchanged // Full YouTube URLs pass through unchanged
assert_eq!( assert_eq!(
expand_shorthand_to_url("https://www.youtube.com/watch?v=UHxw-L2WyyY", &Source::YouTubeVideo), expand_shorthand_to_url(
"https://www.youtube.com/watch?v=UHxw-L2WyyY",
&Source::YouTubeVideo
),
"https://www.youtube.com/watch?v=UHxw-L2WyyY" "https://www.youtube.com/watch?v=UHxw-L2WyyY"
); );
} }
@ -2142,14 +2298,32 @@ mod tests {
expected: Source::YouTubeChannel, expected: Source::YouTubeChannel,
}, },
// Bare video ID — exactly 11 chars [A-Za-z0-9_-] // Bare video ID — exactly 11 chars [A-Za-z0-9_-]
TestCase { url: "yt:dQw4w9WgXcQ", expected: Source::YouTubeVideo }, TestCase {
TestCase { url: "youtube:dQw4w9WgXcQ", expected: Source::YouTubeVideo }, url: "yt:dQw4w9WgXcQ",
TestCase { url: "yt:a_b-c_d-e_4", expected: Source::YouTubeVideo }, expected: Source::YouTubeVideo,
},
TestCase {
url: "youtube:dQw4w9WgXcQ",
expected: Source::YouTubeVideo,
},
TestCase {
url: "yt:a_b-c_d-e_4",
expected: Source::YouTubeVideo,
},
// Non-ID: wrong length (9 chars) → Other // Non-ID: wrong length (9 chars) → Other
TestCase { url: "yt:not-video", expected: Source::Other }, TestCase {
url: "yt:not-video",
expected: Source::Other,
},
// Reserved prefixes still route correctly when segment looks like an ID // Reserved prefixes still route correctly when segment looks like an ID
TestCase { url: "yt:playlist/dQw4w9WgXcQ", expected: Source::YouTubePlaylist }, TestCase {
TestCase { url: "yt:@dQw4w9WgXcQ", expected: Source::YouTubeChannel }, url: "yt:playlist/dQw4w9WgXcQ",
expected: Source::YouTubePlaylist,
},
TestCase {
url: "yt:@dQw4w9WgXcQ",
expected: Source::YouTubeChannel,
},
]; ];
for case in &shorthand_cases { for case in &shorthand_cases {
@ -2165,11 +2339,17 @@ mod tests {
#[test] #[test]
fn test_is_youtube_video_id() { fn test_is_youtube_video_id() {
assert!(is_youtube_video_id("dQw4w9WgXcQ"), "canonical ID"); assert!(is_youtube_video_id("dQw4w9WgXcQ"), "canonical ID");
assert!(is_youtube_video_id("a_b-c_d-e_4"), "11-char ID with _ and -"); assert!(
is_youtube_video_id("a_b-c_d-e_4"),
"11-char ID with _ and -"
);
assert!(!is_youtube_video_id(""), "empty"); assert!(!is_youtube_video_id(""), "empty");
assert!(!is_youtube_video_id("short"), "too short"); assert!(!is_youtube_video_id("short"), "too short");
assert!(!is_youtube_video_id("toolong12345"), "too long (12 chars)"); assert!(!is_youtube_video_id("toolong12345"), "too long (12 chars)");
assert!(!is_youtube_video_id("dQw4w9WgXc!"), "invalid char (! at pos 11)"); assert!(
!is_youtube_video_id("dQw4w9WgXc!"),
"invalid char (! at pos 11)"
);
assert!(!is_youtube_video_id("not-video"), "9 chars — too short"); assert!(!is_youtube_video_id("not-video"), "9 chars — too short");
assert!(!is_youtube_video_id("dQw4w9WgXC!!"), "12 chars + bad char"); assert!(!is_youtube_video_id("dQw4w9WgXC!!"), "12 chars + bad char");
} }
@ -2289,11 +2469,17 @@ mod tests {
// --- expand_shorthand_to_url --- // --- expand_shorthand_to_url ---
assert_eq!( assert_eq!(
expand_shorthand_to_url("spotify:track:4iV5W9uYEdYUVa79Axb7Rh", &Source::SpotifyTrack), expand_shorthand_to_url(
"spotify:track:4iV5W9uYEdYUVa79Axb7Rh",
&Source::SpotifyTrack
),
"https://open.spotify.com/track/4iV5W9uYEdYUVa79Axb7Rh" "https://open.spotify.com/track/4iV5W9uYEdYUVa79Axb7Rh"
); );
assert_eq!( assert_eq!(
expand_shorthand_to_url("spotify:album:1DFixLWuPkv3KT3TnV35m3", &Source::SpotifyAlbum), expand_shorthand_to_url(
"spotify:album:1DFixLWuPkv3KT3TnV35m3",
&Source::SpotifyAlbum
),
"https://open.spotify.com/album/1DFixLWuPkv3KT3TnV35m3" "https://open.spotify.com/album/1DFixLWuPkv3KT3TnV35m3"
); );
assert_eq!( assert_eq!(
@ -2305,9 +2491,18 @@ mod tests {
); );
// --- source_metadata --- // --- source_metadata ---
assert_eq!(source_metadata(Source::SpotifyTrack), ("spotify", "music", "audio")); assert_eq!(
assert_eq!(source_metadata(Source::SpotifyAlbum), ("spotify", "album", "container")); source_metadata(Source::SpotifyTrack),
assert_eq!(source_metadata(Source::SpotifyPlaylist), ("spotify", "playlist", "container")); ("spotify", "music", "audio")
);
assert_eq!(
source_metadata(Source::SpotifyAlbum),
("spotify", "album", "container")
);
assert_eq!(
source_metadata(Source::SpotifyPlaylist),
("spotify", "playlist", "container")
);
} }
#[test] #[test]
@ -2566,25 +2761,37 @@ mod tests {
#[test] #[test]
fn youtube_video_uses_title() { fn youtube_video_uses_title() {
let m = meta(None, Some("How to Rust"), None, None, None); let m = meta(None, Some("How to Rust"), None, None, None);
assert_eq!(generate_entry_title(Source::YouTubeVideo, &m), "How to Rust"); assert_eq!(
generate_entry_title(Source::YouTubeVideo, &m),
"How to Rust"
);
} }
#[test] #[test]
fn youtube_video_fallback() { fn youtube_video_fallback() {
let m = meta(None, None, None, None, None); let m = meta(None, None, None, None, None);
assert_eq!(generate_entry_title(Source::YouTubeVideo, &m), "YouTube Video"); assert_eq!(
generate_entry_title(Source::YouTubeVideo, &m),
"YouTube Video"
);
} }
#[test] #[test]
fn youtube_playlist_uses_title() { fn youtube_playlist_uses_title() {
let m = meta(None, Some("Rust Tutorial Series"), None, None, None); let m = meta(None, Some("Rust Tutorial Series"), None, None, None);
assert_eq!(generate_entry_title(Source::YouTubePlaylist, &m), "Rust Tutorial Series"); assert_eq!(
generate_entry_title(Source::YouTubePlaylist, &m),
"Rust Tutorial Series"
);
} }
#[test] #[test]
fn youtube_channel_uses_author() { fn youtube_channel_uses_author() {
let m = meta(Some("Rust By Example"), None, None, None, None); let m = meta(Some("Rust By Example"), None, None, None, None);
assert_eq!(generate_entry_title(Source::YouTubeChannel, &m), "Archival of Rust By Example"); assert_eq!(
generate_entry_title(Source::YouTubeChannel, &m),
"Archival of Rust By Example"
);
} }
#[test] #[test]
@ -2596,7 +2803,10 @@ mod tests {
#[test] #[test]
fn tweet_uses_excerpt_and_author() { fn tweet_uses_excerpt_and_author() {
let m = meta(Some("alice"), None, Some("Hello world"), None, None); let m = meta(Some("alice"), None, Some("Hello world"), None, None);
assert_eq!(generate_entry_title(Source::Tweet, &m), "Hello world \u{2014} @alice"); assert_eq!(
generate_entry_title(Source::Tweet, &m),
"Hello world \u{2014} @alice"
);
} }
#[test] #[test]
@ -2612,30 +2822,48 @@ mod tests {
#[test] #[test]
fn tweet_thread_uses_author() { fn tweet_thread_uses_author() {
let m = meta(Some("bob"), None, None, None, None); let m = meta(Some("bob"), None, None, None, None);
assert_eq!(generate_entry_title(Source::TweetThread, &m), "Thread by @bob"); assert_eq!(
generate_entry_title(Source::TweetThread, &m),
"Thread by @bob"
);
} }
#[test] #[test]
fn instagram_uses_author() { fn instagram_uses_author() {
let m = meta(Some("photographer"), None, None, None, None); let m = meta(Some("photographer"), None, None, None, None);
assert_eq!(generate_entry_title(Source::Instagram, &m), "Post by @photographer"); assert_eq!(
generate_entry_title(Source::Instagram, &m),
"Post by @photographer"
);
} }
#[test] #[test]
fn facebook_uses_author_no_at() { fn facebook_uses_author_no_at() {
let m = meta(Some("John Doe"), None, None, None, None); let m = meta(Some("John Doe"), None, None, None, None);
assert_eq!(generate_entry_title(Source::Facebook, &m), "Post by John Doe"); assert_eq!(
generate_entry_title(Source::Facebook, &m),
"Post by John Doe"
);
} }
#[test] #[test]
fn tiktok_uses_author() { fn tiktok_uses_author() {
let m = meta(Some("dancemaster"), None, None, None, None); let m = meta(Some("dancemaster"), None, None, None, None);
assert_eq!(generate_entry_title(Source::TikTok, &m), "TikTok by @dancemaster"); assert_eq!(
generate_entry_title(Source::TikTok, &m),
"TikTok by @dancemaster"
);
} }
#[test] #[test]
fn reddit_full_fields() { fn reddit_full_fields() {
let m = meta(None, Some("My first Rust project"), None, Some("rust"), Some("newbie")); let m = meta(
None,
Some("My first Rust project"),
None,
Some("rust"),
Some("newbie"),
);
assert_eq!( assert_eq!(
generate_entry_title(Source::Reddit, &m), generate_entry_title(Source::Reddit, &m),
"My first Rust project \u{2014} r/rust (u/newbie)" "My first Rust project \u{2014} r/rust (u/newbie)"
@ -2645,7 +2873,10 @@ mod tests {
#[test] #[test]
fn snapchat_uses_author() { fn snapchat_uses_author() {
let m = meta(Some("snapuser"), None, None, None, None); let m = meta(Some("snapuser"), None, None, None, None);
assert_eq!(generate_entry_title(Source::Snapchat, &m), "Snap by snapuser"); assert_eq!(
generate_entry_title(Source::Snapchat, &m),
"Snap by snapuser"
);
} }
#[test] #[test]
@ -2668,7 +2899,10 @@ mod tests {
}"#; }"#;
let meta = tweet_metadata_from_json(json); let meta = tweet_metadata_from_json(json);
assert_eq!(meta.author, Some("rustacean".to_string())); assert_eq!(meta.author, Some("rustacean".to_string()));
assert_eq!(meta.caption, Some("Hello Rust world, this is a test tweet".to_string())); assert_eq!(
meta.caption,
Some("Hello Rust world, this is a test tweet".to_string())
);
} }
} }
@ -2735,4 +2969,94 @@ mod tests {
assert_eq!(locator_to_playlist_url("https://example.com/page"), None); assert_eq!(locator_to_playlist_url("https://example.com/page"), None);
} }
mod freedium_supported_url_tests {
use super::is_freedium_supported_url;
#[test]
fn medium_bare_domain() {
assert!(is_freedium_supported_url("https://medium.com/some/article"));
}
#[test]
fn medium_subdomain() {
// Custom Medium publication domains use *.medium.com
assert!(is_freedium_supported_url(
"https://towardsdatascience.medium.com/article-slug-abc123"
));
}
#[test]
fn nytimes() {
assert!(is_freedium_supported_url(
"https://www.nytimes.com/2024/01/01/tech/ai.html"
));
}
#[test]
fn washingtonpost() {
assert!(is_freedium_supported_url(
"https://www.washingtonpost.com/technology/article"
));
}
#[test]
fn bloomberg() {
assert!(is_freedium_supported_url(
"https://bloomberg.com/news/articles/2024-01-01/story"
));
}
#[test]
fn reuters() {
assert!(is_freedium_supported_url(
"https://www.reuters.com/technology/story-2024"
));
}
#[test]
fn economist() {
assert!(is_freedium_supported_url(
"https://www.economist.com/science-and-technology/2024/01/01/article"
));
}
#[test]
fn financial_times() {
assert!(is_freedium_supported_url(
"https://www.ft.com/content/some-uuid"
));
}
#[test]
fn non_paywall_site_rejected() {
// The original bug: borretti.me was incorrectly routed through Freedium
assert!(!is_freedium_supported_url(
"https://borretti.me/article/notes-on-managing-adhd"
));
}
#[test]
fn generic_url_rejected() {
assert!(!is_freedium_supported_url("https://example.com/page"));
}
#[test]
fn freedium_mirror_itself_rejected() {
// Existing check already blocks mirror re-wrapping, but belt-and-suspenders
assert!(!is_freedium_supported_url(
"https://freedium-mirror.cfd/https://medium.com/article"
));
}
#[test]
fn unparseable_url_rejected() {
assert!(!is_freedium_supported_url("not a url at all"));
}
#[test]
fn lookalike_subdomain_rejected() {
// "notmedium.com" should not match due to the dot-prefix check
assert!(!is_freedium_supported_url("https://notmedium.com/article"));
}
}
} }