mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 12:55:00 +02:00
core: gate Freedium routing to supported paywall domains only
Previously, any WebPage capture with via_freedium=true was routed through the Freedium mirror regardless of the URL's host. This caused non-paywall sites like borretti.me to be fetched through Freedium, which is wrong — Freedium only knows how to handle a specific set of publications. Add is_freedium_supported_url() with a static allowlist of the 7 hosts Freedium explicitly supports per its homepage announcement: Medium, NYT, WaPo, Bloomberg, Reuters, Economist, Financial Times The gate matches on the bare domain or any subdomain (e.g. towardsdatascience.medium.com). Unparseable URLs fall through to a direct fetch. The existing freedium-mirror.cfd re-wrap guard is kept as a belt-and-suspenders check after the new allowlist test. Fixes: https://borretti.me/article/notes-on-managing-adhd archived via Freedium despite not being a Medium article.
This commit is contained in:
parent
6a0f59ea94
commit
1342ff1dca
1 changed files with 500 additions and 176 deletions
|
|
@ -1,13 +1,17 @@
|
||||||
|
use crate::{
|
||||||
|
archive::{self, ArchivePaths},
|
||||||
|
database, downloader,
|
||||||
|
twitter::parse_tweet_id,
|
||||||
|
};
|
||||||
use anyhow::{Context, Result};
|
use anyhow::{Context, Result};
|
||||||
use chrono::Local;
|
use chrono::Local;
|
||||||
use uuid::Uuid;
|
|
||||||
use serde_json::json;
|
use serde_json::json;
|
||||||
use std::{
|
use std::{
|
||||||
collections::{HashMap, HashSet},
|
collections::{HashMap, HashSet},
|
||||||
fs,
|
fs,
|
||||||
path::{Path, PathBuf},
|
path::{Path, PathBuf},
|
||||||
};
|
};
|
||||||
use crate::{archive::{self, ArchivePaths}, database, downloader, twitter::parse_tweet_id};
|
use uuid::Uuid;
|
||||||
|
|
||||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||||
pub enum Source {
|
pub enum Source {
|
||||||
|
|
@ -125,15 +129,12 @@ pub fn resolve_cookies_for_url(
|
||||||
None => true,
|
None => true,
|
||||||
Some(pattern) => match rule.pattern_kind.as_str() {
|
Some(pattern) => match rule.pattern_kind.as_str() {
|
||||||
"wildcard" => wildcard_matches(pattern, url),
|
"wildcard" => wildcard_matches(pattern, url),
|
||||||
"regex" => regex::Regex::new(pattern)
|
"regex" => regex::Regex::new(pattern).is_ok_and(|re| re.is_match(url)),
|
||||||
.is_ok_and(|re| re.is_match(url)),
|
|
||||||
_ => false,
|
_ => false,
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
if applies {
|
if applies {
|
||||||
if let Ok(map) =
|
if let Ok(map) = serde_json::from_str::<HashMap<String, String>>(&rule.cookies_json) {
|
||||||
serde_json::from_str::<HashMap<String, String>>(&rule.cookies_json)
|
|
||||||
{
|
|
||||||
result.extend(map);
|
result.extend(map);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -163,7 +164,10 @@ fn wildcard_matches(pattern: &str, url: &str) -> bool {
|
||||||
match ch {
|
match ch {
|
||||||
'*' => pat.push_str(".*"),
|
'*' => pat.push_str(".*"),
|
||||||
'?' => pat.push('.'),
|
'?' => pat.push('.'),
|
||||||
c if "$.+[]{}()|^\\".contains(c) => { pat.push('\\'); pat.push(c); }
|
c if "$.+[]{}()|^\\".contains(c) => {
|
||||||
|
pat.push('\\');
|
||||||
|
pat.push(c);
|
||||||
|
}
|
||||||
c => pat.push(c),
|
c => pat.push(c),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -171,10 +175,48 @@ fn wildcard_matches(pattern: &str, url: &str) -> bool {
|
||||||
regex::Regex::new(&pat).is_ok_and(|re| re.is_match(&target))
|
regex::Regex::new(&pat).is_ok_and(|re| re.is_match(&target))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The set of base domains that Freedium supports as of its "7 new sources" announcement.
|
||||||
|
/// A URL matches if its hostname is exactly `domain` or any subdomain (`*.domain`).
|
||||||
|
///
|
||||||
|
/// Source: https://freedium-mirror.cfd/ — "7 new sources supported! Medium, NYT, WaPo,
|
||||||
|
/// Bloomberg, Reuters, Economist, and Financial Times."
|
||||||
|
const FREEDIUM_SUPPORTED_HOSTS: &[&str] = &[
|
||||||
|
"medium.com",
|
||||||
|
"nytimes.com",
|
||||||
|
"washingtonpost.com",
|
||||||
|
"bloomberg.com",
|
||||||
|
"reuters.com",
|
||||||
|
"economist.com",
|
||||||
|
"ft.com",
|
||||||
|
];
|
||||||
|
|
||||||
|
/// Returns true when `url`'s hostname is one of the domains Freedium explicitly supports,
|
||||||
|
/// either as the bare domain or a subdomain (e.g. `towardsdatascience.medium.com`).
|
||||||
|
/// Unparseable URLs return false so we never accidentally proxy an unknown host.
|
||||||
|
fn is_freedium_supported_url(url: &str) -> bool {
|
||||||
|
use reqwest::Url as ReqwestUrl;
|
||||||
|
let host = match ReqwestUrl::parse(url)
|
||||||
|
.ok()
|
||||||
|
.and_then(|u| u.host_str().map(str::to_string))
|
||||||
|
{
|
||||||
|
Some(h) => h,
|
||||||
|
None => return false,
|
||||||
|
};
|
||||||
|
FREEDIUM_SUPPORTED_HOSTS
|
||||||
|
.iter()
|
||||||
|
.any(|&domain| host == domain || host.ends_with(&format!(".{domain}")))
|
||||||
|
}
|
||||||
|
|
||||||
fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
|
fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
|
||||||
match source {
|
match source {
|
||||||
Source::YouTubeVideo => meta.title.clone().unwrap_or_else(|| "YouTube Video".to_string()),
|
Source::YouTubeVideo => meta
|
||||||
Source::YouTubePlaylist => meta.title.clone().unwrap_or_else(|| "YouTube Playlist".to_string()),
|
.title
|
||||||
|
.clone()
|
||||||
|
.unwrap_or_else(|| "YouTube Video".to_string()),
|
||||||
|
Source::YouTubePlaylist => meta
|
||||||
|
.title
|
||||||
|
.clone()
|
||||||
|
.unwrap_or_else(|| "YouTube Playlist".to_string()),
|
||||||
Source::YouTubeChannel => format!(
|
Source::YouTubeChannel => format!(
|
||||||
"Archival of {}",
|
"Archival of {}",
|
||||||
meta.author.as_deref().unwrap_or("Unknown Channel")
|
meta.author.as_deref().unwrap_or("Unknown Channel")
|
||||||
|
|
@ -186,18 +228,28 @@ fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
|
||||||
None => title.to_string(),
|
None => title.to_string(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Source::YouTubeMusicPlaylist => {
|
Source::YouTubeMusicPlaylist => meta
|
||||||
meta.title.clone().unwrap_or_else(|| "YouTube Music Playlist".to_string())
|
.title
|
||||||
}
|
.clone()
|
||||||
Source::SpotifyTrack | Source::SpotifyAlbum | Source::SpotifyPlaylist => {
|
.unwrap_or_else(|| "YouTube Music Playlist".to_string()),
|
||||||
meta.title.clone().unwrap_or_else(|| "Spotify Content".to_string())
|
Source::SpotifyTrack | Source::SpotifyAlbum | Source::SpotifyPlaylist => meta
|
||||||
}
|
.title
|
||||||
|
.clone()
|
||||||
|
.unwrap_or_else(|| "Spotify Content".to_string()),
|
||||||
Source::X => format!("X Media by {}", meta.author.as_deref().unwrap_or("unknown")),
|
Source::X => format!("X Media by {}", meta.author.as_deref().unwrap_or("unknown")),
|
||||||
Source::Tweet => {
|
Source::Tweet => {
|
||||||
let excerpt = meta.caption_excerpt().unwrap_or_else(|| "Tweet".to_string());
|
let excerpt = meta
|
||||||
format!("{} \u{2014} @{}", excerpt, meta.author.as_deref().unwrap_or("unknown"))
|
.caption_excerpt()
|
||||||
|
.unwrap_or_else(|| "Tweet".to_string());
|
||||||
|
format!(
|
||||||
|
"{} \u{2014} @{}",
|
||||||
|
excerpt,
|
||||||
|
meta.author.as_deref().unwrap_or("unknown")
|
||||||
|
)
|
||||||
|
}
|
||||||
|
Source::TweetThread => {
|
||||||
|
format!("Thread by @{}", meta.author.as_deref().unwrap_or("unknown"))
|
||||||
}
|
}
|
||||||
Source::TweetThread => format!("Thread by @{}", meta.author.as_deref().unwrap_or("unknown")),
|
|
||||||
Source::Instagram => format!("Post by @{}", meta.author.as_deref().unwrap_or("unknown")),
|
Source::Instagram => format!("Post by @{}", meta.author.as_deref().unwrap_or("unknown")),
|
||||||
Source::Facebook => format!("Post by {}", meta.author.as_deref().unwrap_or("unknown")),
|
Source::Facebook => format!("Post by {}", meta.author.as_deref().unwrap_or("unknown")),
|
||||||
Source::TikTok => format!("TikTok by @{}", meta.author.as_deref().unwrap_or("unknown")),
|
Source::TikTok => format!("TikTok by @{}", meta.author.as_deref().unwrap_or("unknown")),
|
||||||
|
|
@ -208,23 +260,39 @@ fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
|
||||||
meta.post_author.as_deref().unwrap_or("unknown")
|
meta.post_author.as_deref().unwrap_or("unknown")
|
||||||
),
|
),
|
||||||
Source::Snapchat => format!("Snap by {}", meta.author.as_deref().unwrap_or("unknown")),
|
Source::Snapchat => format!("Snap by {}", meta.author.as_deref().unwrap_or("unknown")),
|
||||||
Source::Local => meta.title.clone().unwrap_or_else(|| "Local File".to_string()),
|
Source::Local => meta
|
||||||
Source::Url => meta.title.clone().unwrap_or_else(|| "Downloaded File".to_string()),
|
.title
|
||||||
Source::WebPage => meta.title.clone().unwrap_or_else(|| "Archived Web Page".to_string()),
|
.clone()
|
||||||
|
.unwrap_or_else(|| "Local File".to_string()),
|
||||||
|
Source::Url => meta
|
||||||
|
.title
|
||||||
|
.clone()
|
||||||
|
.unwrap_or_else(|| "Downloaded File".to_string()),
|
||||||
|
Source::WebPage => meta
|
||||||
|
.title
|
||||||
|
.clone()
|
||||||
|
.unwrap_or_else(|| "Archived Web Page".to_string()),
|
||||||
Source::Other => "Archived Content".to_string(),
|
Source::Other => "Archived Content".to_string(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
/// Returns true when `s` is a valid bare YouTube video ID:
|
/// Returns true when `s` is a valid bare YouTube video ID:
|
||||||
/// exactly 11 characters from the set `[A-Za-z0-9_-]`.
|
/// exactly 11 characters from the set `[A-Za-z0-9_-]`.
|
||||||
fn is_youtube_video_id(s: &str) -> bool {
|
fn is_youtube_video_id(s: &str) -> bool {
|
||||||
s.len() == 11 && s.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'_' || b == b'-')
|
s.len() == 11
|
||||||
|
&& s.bytes()
|
||||||
|
.all(|b| b.is_ascii_alphanumeric() || b == b'_' || b == b'-')
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
fn expand_shorthand_to_url(path: &str, source: &Source) -> String {
|
fn expand_shorthand_to_url(path: &str, source: &Source) -> String {
|
||||||
// YouTube shorthands: yt:video/ID, yt:playlist/ID, yt:@handle, yt:channel/ID, etc.
|
// YouTube shorthands: yt:video/ID, yt:playlist/ID, yt:@handle, yt:channel/ID, etc.
|
||||||
if matches!(source, Source::YouTubeVideo | Source::YouTubePlaylist | Source::YouTubeChannel) {
|
if matches!(
|
||||||
if let Some(after) = path.strip_prefix("yt:").or_else(|| path.strip_prefix("youtube:")) {
|
source,
|
||||||
|
Source::YouTubeVideo | Source::YouTubePlaylist | Source::YouTubeChannel
|
||||||
|
) {
|
||||||
|
if let Some(after) = path
|
||||||
|
.strip_prefix("yt:")
|
||||||
|
.or_else(|| path.strip_prefix("youtube:"))
|
||||||
|
{
|
||||||
if let Some(id) = after
|
if let Some(id) = after
|
||||||
.strip_prefix("video/")
|
.strip_prefix("video/")
|
||||||
.or_else(|| after.strip_prefix("short/"))
|
.or_else(|| after.strip_prefix("short/"))
|
||||||
|
|
@ -255,7 +323,10 @@ fn expand_shorthand_to_url(path: &str, source: &Source) -> String {
|
||||||
}
|
}
|
||||||
|
|
||||||
// YouTube Music shorthands: ytm:ID (track) or ytm:playlist/ID
|
// YouTube Music shorthands: ytm:ID (track) or ytm:playlist/ID
|
||||||
if matches!(source, Source::YouTubeMusicTrack | Source::YouTubeMusicPlaylist) {
|
if matches!(
|
||||||
|
source,
|
||||||
|
Source::YouTubeMusicTrack | Source::YouTubeMusicPlaylist
|
||||||
|
) {
|
||||||
if let Some(after) = path.strip_prefix("ytm:") {
|
if let Some(after) = path.strip_prefix("ytm:") {
|
||||||
if let Some(id) = after.strip_prefix("playlist/") {
|
if let Some(id) = after.strip_prefix("playlist/") {
|
||||||
return format!("https://music.youtube.com/playlist?list={id}");
|
return format!("https://music.youtube.com/playlist?list={id}");
|
||||||
|
|
@ -266,7 +337,10 @@ fn expand_shorthand_to_url(path: &str, source: &Source) -> String {
|
||||||
}
|
}
|
||||||
|
|
||||||
// Spotify shorthands: spotify:track:ID, spotify:album:ID, spotify:playlist:ID
|
// Spotify shorthands: spotify:track:ID, spotify:album:ID, spotify:playlist:ID
|
||||||
if matches!(source, Source::SpotifyTrack | Source::SpotifyAlbum | Source::SpotifyPlaylist) {
|
if matches!(
|
||||||
|
source,
|
||||||
|
Source::SpotifyTrack | Source::SpotifyAlbum | Source::SpotifyPlaylist
|
||||||
|
) {
|
||||||
if let Some(after) = path.strip_prefix("spotify:") {
|
if let Some(after) = path.strip_prefix("spotify:") {
|
||||||
if let Some(id) = after.strip_prefix("track:") {
|
if let Some(id) = after.strip_prefix("track:") {
|
||||||
return format!("https://open.spotify.com/track/{id}");
|
return format!("https://open.spotify.com/track/{id}");
|
||||||
|
|
@ -869,8 +943,12 @@ fn register_tweet_artifacts(
|
||||||
},
|
},
|
||||||
)?;
|
)?;
|
||||||
let json_path = store_path.join(relpath);
|
let json_path = store_path.join(relpath);
|
||||||
let json_str = fs::read_to_string(&json_path)
|
let json_str = fs::read_to_string(&json_path).with_context(|| {
|
||||||
.with_context(|| format!("failed to read tweet JSON for artifact registration: {}", json_path.display()))?;
|
format!(
|
||||||
|
"failed to read tweet JSON for artifact registration: {}",
|
||||||
|
json_path.display()
|
||||||
|
)
|
||||||
|
})?;
|
||||||
for (role, raw_relpath) in tweet_raw_artifacts(&json_str)? {
|
for (role, raw_relpath) in tweet_raw_artifacts(&json_str)? {
|
||||||
let raw_path = PathBuf::from(&raw_relpath);
|
let raw_path = PathBuf::from(&raw_relpath);
|
||||||
let blob = blob_record_for_raw_relpath(store_path, &raw_path)?;
|
let blob = blob_record_for_raw_relpath(store_path, &raw_path)?;
|
||||||
|
|
@ -1081,10 +1159,12 @@ pub fn perform_capture(
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
let container_title = playlist_info
|
let container_title = playlist_info.title.clone().or_else(|| {
|
||||||
.title
|
playlist_info
|
||||||
|
.uploader
|
||||||
.clone()
|
.clone()
|
||||||
.or_else(|| playlist_info.uploader.clone().map(|u| format!("{u} (channel)")));
|
.map(|u| format!("{u} (channel)"))
|
||||||
|
});
|
||||||
|
|
||||||
// Sync mode: reuse an existing container so we don't create a duplicate
|
// Sync mode: reuse an existing container so we don't create a duplicate
|
||||||
// root entry on every sync run. Non-sync always creates a fresh container.
|
// root entry on every sync run. Non-sync always creates a fresh container.
|
||||||
|
|
@ -1092,55 +1172,89 @@ pub fn perform_capture(
|
||||||
match archive::find_container_entry_id_by_canonical_url(&conn, &canonical_url) {
|
match archive::find_container_entry_id_by_canonical_url(&conn, &canonical_url) {
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
return Err(fail_run(
|
return Err(fail_run(
|
||||||
&conn, &run, &item,
|
&conn,
|
||||||
|
&run,
|
||||||
|
&item,
|
||||||
&format!("Failed to query existing container: {e:#}"),
|
&format!("Failed to query existing container: {e:#}"),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
Ok(Some(existing_id)) => {
|
Ok(Some(existing_id)) => {
|
||||||
// Container already exists — mark the run item done pointing at it;
|
// Container already exists — mark the run item done pointing at it;
|
||||||
// don't call record_container_entry (that would create a duplicate).
|
// don't call record_container_entry (that would create a duplicate).
|
||||||
if let Err(e) = database::complete_archive_run_item(&conn, item.id, existing_id) {
|
if let Err(e) = database::complete_archive_run_item(&conn, item.id, existing_id)
|
||||||
|
{
|
||||||
return Err(fail_run(
|
return Err(fail_run(
|
||||||
&conn, &run, &item,
|
&conn,
|
||||||
|
&run,
|
||||||
|
&item,
|
||||||
&format!("Failed to complete run item for existing container: {e:#}"),
|
&format!("Failed to complete run item for existing container: {e:#}"),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
let archived = match archive::get_archived_playlist_child_urls(&conn, &canonical_url) {
|
let archived =
|
||||||
|
match archive::get_archived_playlist_child_urls(&conn, &canonical_url) {
|
||||||
Ok(set) => set,
|
Ok(set) => set,
|
||||||
Err(e) => return Err(fail_run(
|
Err(e) => {
|
||||||
&conn, &run, &item,
|
return Err(fail_run(
|
||||||
|
&conn,
|
||||||
|
&run,
|
||||||
|
&item,
|
||||||
&format!("Failed to query archived playlist children: {e:#}"),
|
&format!("Failed to query archived playlist children: {e:#}"),
|
||||||
)),
|
));
|
||||||
|
}
|
||||||
};
|
};
|
||||||
(existing_id, archived)
|
(existing_id, archived)
|
||||||
}
|
}
|
||||||
Ok(None) => {
|
Ok(None) => {
|
||||||
// First sync run for this playlist — create the container normally.
|
// First sync run for this playlist — create the container normally.
|
||||||
let e = match record_container_entry(
|
let e = match record_container_entry(
|
||||||
&conn, store_path, user_id, &run, &item, locator, &canonical_url,
|
&conn,
|
||||||
source, container_title, &playlist_info.playlist_id,
|
store_path,
|
||||||
|
user_id,
|
||||||
|
&run,
|
||||||
|
&item,
|
||||||
|
locator,
|
||||||
|
&canonical_url,
|
||||||
|
source,
|
||||||
|
container_title,
|
||||||
|
&playlist_info.playlist_id,
|
||||||
playlist_info.uploader.as_deref(),
|
playlist_info.uploader.as_deref(),
|
||||||
) {
|
) {
|
||||||
Ok(e) => e,
|
Ok(e) => e,
|
||||||
Err(e) => return Err(fail_run(
|
Err(e) => {
|
||||||
&conn, &run, &item,
|
return Err(fail_run(
|
||||||
|
&conn,
|
||||||
|
&run,
|
||||||
|
&item,
|
||||||
&format!("Failed to create container entry: {e:#}"),
|
&format!("Failed to create container entry: {e:#}"),
|
||||||
)),
|
));
|
||||||
|
}
|
||||||
};
|
};
|
||||||
(e.id, std::collections::HashSet::new())
|
(e.id, std::collections::HashSet::new())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
let e = match record_container_entry(
|
let e = match record_container_entry(
|
||||||
&conn, store_path, user_id, &run, &item, locator, &canonical_url,
|
&conn,
|
||||||
source, container_title, &playlist_info.playlist_id,
|
store_path,
|
||||||
|
user_id,
|
||||||
|
&run,
|
||||||
|
&item,
|
||||||
|
locator,
|
||||||
|
&canonical_url,
|
||||||
|
source,
|
||||||
|
container_title,
|
||||||
|
&playlist_info.playlist_id,
|
||||||
playlist_info.uploader.as_deref(),
|
playlist_info.uploader.as_deref(),
|
||||||
) {
|
) {
|
||||||
Ok(e) => e,
|
Ok(e) => e,
|
||||||
Err(e) => return Err(fail_run(
|
Err(e) => {
|
||||||
&conn, &run, &item,
|
return Err(fail_run(
|
||||||
|
&conn,
|
||||||
|
&run,
|
||||||
|
&item,
|
||||||
&format!("Failed to create container entry: {e:#}"),
|
&format!("Failed to create container entry: {e:#}"),
|
||||||
)),
|
));
|
||||||
|
}
|
||||||
};
|
};
|
||||||
(e.id, std::collections::HashSet::new())
|
(e.id, std::collections::HashSet::new())
|
||||||
};
|
};
|
||||||
|
|
@ -1168,7 +1282,6 @@ pub fn perform_capture(
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
let child_timestamp = format!(
|
let child_timestamp = format!(
|
||||||
"{}-{}",
|
"{}-{}",
|
||||||
Local::now().format("%Y-%m-%dT%H-%M-%S%.3f"),
|
Local::now().format("%Y-%m-%dT%H-%M-%S%.3f"),
|
||||||
|
|
@ -1187,7 +1300,10 @@ pub fn perform_capture(
|
||||||
) {
|
) {
|
||||||
Ok(i) => i,
|
Ok(i) => i,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("warn: playlist item {} create_run_item failed: {e:#}", playlist_item.url);
|
eprintln!(
|
||||||
|
"warn: playlist item {} create_run_item failed: {e:#}",
|
||||||
|
playlist_item.url
|
||||||
|
);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
@ -1241,7 +1357,8 @@ pub fn perform_capture(
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("warn: stat child temp file: {e:#}");
|
eprintln!("warn: stat child temp file: {e:#}");
|
||||||
let _ = database::fail_archive_run_item(
|
let _ = database::fail_archive_run_item(
|
||||||
&conn, child_item.id,
|
&conn,
|
||||||
|
child_item.id,
|
||||||
&format!("failed to stat downloaded file: {e:#}"),
|
&format!("failed to stat downloaded file: {e:#}"),
|
||||||
);
|
);
|
||||||
continue;
|
continue;
|
||||||
|
|
@ -1251,7 +1368,8 @@ pub fn perform_capture(
|
||||||
if let Err(e) = move_temp_to_raw(&temp_file, &hash, store_path) {
|
if let Err(e) = move_temp_to_raw(&temp_file, &hash, store_path) {
|
||||||
eprintln!("warn: move_temp_to_raw child: {e:#}");
|
eprintln!("warn: move_temp_to_raw child: {e:#}");
|
||||||
let _ = database::fail_archive_run_item(
|
let _ = database::fail_archive_run_item(
|
||||||
&conn, child_item.id,
|
&conn,
|
||||||
|
child_item.id,
|
||||||
&format!("failed to move downloaded file: {e:#}"),
|
&format!("failed to move downloaded file: {e:#}"),
|
||||||
);
|
);
|
||||||
continue;
|
continue;
|
||||||
|
|
@ -1286,7 +1404,8 @@ pub fn perform_capture(
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("warn: record child entry: {e:#}");
|
eprintln!("warn: record child entry: {e:#}");
|
||||||
let _ = database::fail_archive_run_item(
|
let _ = database::fail_archive_run_item(
|
||||||
&conn, child_item.id,
|
&conn,
|
||||||
|
child_item.id,
|
||||||
&format!("failed to record entry: {e:#}"),
|
&format!("failed to record entry: {e:#}"),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
@ -1294,9 +1413,13 @@ pub fn perform_capture(
|
||||||
}
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
let _ = fs::remove_dir_all(store_path.join("temp").join(&child_timestamp));
|
let _ = fs::remove_dir_all(store_path.join("temp").join(&child_timestamp));
|
||||||
eprintln!("warn: yt-dlp child download failed for {}: {e:#}", playlist_item.url);
|
eprintln!(
|
||||||
|
"warn: yt-dlp child download failed for {}: {e:#}",
|
||||||
|
playlist_item.url
|
||||||
|
);
|
||||||
let _ = database::fail_archive_run_item(
|
let _ = database::fail_archive_run_item(
|
||||||
&conn, child_item.id,
|
&conn,
|
||||||
|
child_item.id,
|
||||||
&format!("yt-dlp download failed: {e:#}"),
|
&format!("yt-dlp download failed: {e:#}"),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
@ -1313,8 +1436,8 @@ pub fn perform_capture(
|
||||||
|row| row.get(0),
|
|row| row.get(0),
|
||||||
)
|
)
|
||||||
.unwrap_or_else(|_| "completed".to_string());
|
.unwrap_or_else(|_| "completed".to_string());
|
||||||
let completed_child_count: i64 = database::get_run_completed_child_count(&conn, run.id)
|
let completed_child_count: i64 =
|
||||||
.unwrap_or(0);
|
database::get_run_completed_child_count(&conn, run.id).unwrap_or(0);
|
||||||
|
|
||||||
return Ok(CaptureResult {
|
return Ok(CaptureResult {
|
||||||
run_uid: run.run_uid.clone(),
|
run_uid: run.run_uid.clone(),
|
||||||
|
|
@ -1391,20 +1514,36 @@ pub fn perform_capture(
|
||||||
|
|
||||||
// Source: web page — archive as a self-contained HTML snapshot via single-file-cli
|
// Source: web page — archive as a self-contained HTML snapshot via single-file-cli
|
||||||
if source == Source::WebPage {
|
if source == Source::WebPage {
|
||||||
// When via_freedium is enabled and the URL is not already a freedium mirror,
|
// When via_freedium is enabled, the URL is on a Freedium-supported host, and it is
|
||||||
// fetch through the mirror to bypass paywalls. Store the original locator in DB.
|
// not already a freedium mirror URL, fetch through the mirror to bypass paywalls.
|
||||||
|
// Store the original locator in the DB; only the fetch URL changes.
|
||||||
// Use an empty cookie jar for the mirror URL: cookies resolved for the original
|
// Use an empty cookie jar for the mirror URL: cookies resolved for the original
|
||||||
// domain (e.g. NYT, Medium) must not be sent to freedium-mirror.cfd.
|
// domain (e.g. NYT, Medium) must not be sent to freedium-mirror.cfd.
|
||||||
let (fetch_url, fetch_cookies): (String, HashMap<String, String>) =
|
let (fetch_url, fetch_cookies): (String, HashMap<String, String>) = if config.via_freedium
|
||||||
if config.via_freedium && !locator.starts_with("https://freedium-mirror.cfd/") {
|
&& is_freedium_supported_url(locator)
|
||||||
(format!("https://freedium-mirror.cfd/{}", locator), HashMap::new())
|
&& !locator.starts_with("https://freedium-mirror.cfd/")
|
||||||
|
{
|
||||||
|
(
|
||||||
|
format!("https://freedium-mirror.cfd/{}", locator),
|
||||||
|
HashMap::new(),
|
||||||
|
)
|
||||||
} else {
|
} else {
|
||||||
(locator.to_string(), cookies.clone())
|
(locator.to_string(), cookies.clone())
|
||||||
};
|
};
|
||||||
// Key cleanup/title-stripping off the actual fetch host so that
|
// Key cleanup/title-stripping off the actual fetch host so that
|
||||||
// user-supplied freedium-mirror.cfd URLs are also handled correctly.
|
// user-supplied freedium-mirror.cfd URLs are also handled correctly.
|
||||||
let is_freedium_fetch = fetch_url.starts_with("https://freedium-mirror.cfd/");
|
let is_freedium_fetch = fetch_url.starts_with("https://freedium-mirror.cfd/");
|
||||||
match downloader::singlefile::save(&fetch_url, store_path, ×tamp, &fetch_cookies, config.ublock_enabled, config.cookie_ext_enabled, config.reader_mode, config.modal_closer_enabled, is_freedium_fetch) {
|
match downloader::singlefile::save(
|
||||||
|
&fetch_url,
|
||||||
|
store_path,
|
||||||
|
×tamp,
|
||||||
|
&fetch_cookies,
|
||||||
|
config.ublock_enabled,
|
||||||
|
config.cookie_ext_enabled,
|
||||||
|
config.reader_mode,
|
||||||
|
config.modal_closer_enabled,
|
||||||
|
is_freedium_fetch,
|
||||||
|
) {
|
||||||
Ok(result) => {
|
Ok(result) => {
|
||||||
let file_extension = ".html".to_string();
|
let file_extension = ".html".to_string();
|
||||||
let temp_html = store_path
|
let temp_html = store_path
|
||||||
|
|
@ -1415,8 +1554,9 @@ pub fn perform_capture(
|
||||||
// Font extraction: rewrite the HTML in-place before hashing.
|
// Font extraction: rewrite the HTML in-place before hashing.
|
||||||
// Only runs when archive_id is known (server context). CLI passes
|
// Only runs when archive_id is known (server context). CLI passes
|
||||||
// None and keeps fonts embedded — no behaviour change for CLI.
|
// None and keeps fonts embedded — no behaviour change for CLI.
|
||||||
let (html_hash, byte_size, extracted_fonts, html_title) =
|
let (html_hash, byte_size, extracted_fonts, html_title) = if let Some(aid) =
|
||||||
if let Some(aid) = archive_id {
|
archive_id
|
||||||
|
{
|
||||||
let content = fs::read_to_string(&temp_html)
|
let content = fs::read_to_string(&temp_html)
|
||||||
.with_context(|| format!("failed to read {}", temp_html.display()))?;
|
.with_context(|| format!("failed to read {}", temp_html.display()))?;
|
||||||
let (rewritten, fonts) =
|
let (rewritten, fonts) =
|
||||||
|
|
@ -1448,9 +1588,12 @@ pub fn perform_capture(
|
||||||
let fav_hash = result.favicon_hash.as_deref()?;
|
let fav_hash = result.favicon_hash.as_deref()?;
|
||||||
let fav_ext = result.favicon_ext.as_deref()?;
|
let fav_ext = result.favicon_ext.as_deref()?;
|
||||||
let fav_temp = store_path
|
let fav_temp = store_path
|
||||||
.join("temp").join(×tamp)
|
.join("temp")
|
||||||
|
.join(×tamp)
|
||||||
.join(format!("{timestamp}.favicon{fav_ext}"));
|
.join(format!("{timestamp}.favicon{fav_ext}"));
|
||||||
if !fav_temp.exists() { return None; }
|
if !fav_temp.exists() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
let fav_size = fs::metadata(&fav_temp).ok()?.len() as i64;
|
let fav_size = fs::metadata(&fav_temp).ok()?.len() as i64;
|
||||||
let fav_raw = raw_relative_path_from_hash(fav_hash, fav_ext).ok()?;
|
let fav_raw = raw_relative_path_from_hash(fav_hash, fav_ext).ok()?;
|
||||||
if !hash_exists(fav_hash, fav_ext, store_path).ok()? {
|
if !hash_exists(fav_hash, fav_ext, store_path).ok()? {
|
||||||
|
|
@ -1658,7 +1801,13 @@ pub fn perform_capture(
|
||||||
| Source::TikTok
|
| Source::TikTok
|
||||||
| Source::Reddit
|
| Source::Reddit
|
||||||
| Source::Snapchat => {
|
| Source::Snapchat => {
|
||||||
match downloader::ytdlp::download(path.clone(), store_path, ×tamp, quality, &cookies) {
|
match downloader::ytdlp::download(
|
||||||
|
path.clone(),
|
||||||
|
store_path,
|
||||||
|
×tamp,
|
||||||
|
quality,
|
||||||
|
&cookies,
|
||||||
|
) {
|
||||||
Ok(result) => result,
|
Ok(result) => result,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
return Err(fail_run(
|
return Err(fail_run(
|
||||||
|
|
@ -1672,7 +1821,13 @@ pub fn perform_capture(
|
||||||
}
|
}
|
||||||
Source::YouTubeMusicTrack | Source::SpotifyTrack => {
|
Source::YouTubeMusicTrack | Source::SpotifyTrack => {
|
||||||
// Music tracks are always audio-only regardless of the caller's quality hint.
|
// Music tracks are always audio-only regardless of the caller's quality hint.
|
||||||
match downloader::ytdlp::download(path.clone(), store_path, ×tamp, Some("audio"), &cookies) {
|
match downloader::ytdlp::download(
|
||||||
|
path.clone(),
|
||||||
|
store_path,
|
||||||
|
×tamp,
|
||||||
|
Some("audio"),
|
||||||
|
&cookies,
|
||||||
|
) {
|
||||||
Ok(result) => result,
|
Ok(result) => result,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
return Err(fail_run(
|
return Err(fail_run(
|
||||||
|
|
@ -1684,8 +1839,7 @@ pub fn perform_capture(
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Source::Local => {
|
Source::Local => match downloader::local::save(path.clone(), store_path, ×tamp) {
|
||||||
match downloader::local::save(path.clone(), store_path, ×tamp) {
|
|
||||||
Ok(h) => (h, local_file_extension(&path)),
|
Ok(h) => (h, local_file_extension(&path)),
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
return Err(fail_run(
|
return Err(fail_run(
|
||||||
|
|
@ -1695,8 +1849,7 @@ pub fn perform_capture(
|
||||||
&format!("Failed to archive local file: {e}"),
|
&format!("Failed to archive local file: {e}"),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
}
|
},
|
||||||
}
|
|
||||||
Source::YouTubePlaylist | Source::YouTubeChannel => unreachable!(),
|
Source::YouTubePlaylist | Source::YouTubeChannel => unreachable!(),
|
||||||
_ => unreachable!(),
|
_ => unreachable!(),
|
||||||
};
|
};
|
||||||
|
|
@ -1811,10 +1964,7 @@ pub fn perform_rearchive(
|
||||||
// Parse source_metadata_json for tweet_id and requested_locator.
|
// Parse source_metadata_json for tweet_id and requested_locator.
|
||||||
let meta: serde_json::Value = serde_json::from_str(&entry.source_metadata_json)
|
let meta: serde_json::Value = serde_json::from_str(&entry.source_metadata_json)
|
||||||
.unwrap_or(serde_json::Value::Object(Default::default()));
|
.unwrap_or(serde_json::Value::Object(Default::default()));
|
||||||
let requested_locator = meta["requested_locator"]
|
let requested_locator = meta["requested_locator"].as_str().unwrap_or("").to_string();
|
||||||
.as_str()
|
|
||||||
.unwrap_or("")
|
|
||||||
.to_string();
|
|
||||||
let tweet_id = meta["tweet_id"].as_str().unwrap_or("").to_string();
|
let tweet_id = meta["tweet_id"].as_str().unwrap_or("").to_string();
|
||||||
if requested_locator.is_empty() || tweet_id.is_empty() {
|
if requested_locator.is_empty() || tweet_id.is_empty() {
|
||||||
return Ok(RearchiveResult {
|
return Ok(RearchiveResult {
|
||||||
|
|
@ -1858,7 +2008,10 @@ pub fn perform_rearchive(
|
||||||
|
|
||||||
database::refresh_entry_cached_bytes(&conn, entry.id)?;
|
database::refresh_entry_cached_bytes(&conn, entry.id)?;
|
||||||
|
|
||||||
eprintln!("info: rearchived entry {entry_uid}: {} tweet JSONs", tweet_json_relpaths.len());
|
eprintln!(
|
||||||
|
"info: rearchived entry {entry_uid}: {} tweet JSONs",
|
||||||
|
tweet_json_relpaths.len()
|
||||||
|
);
|
||||||
|
|
||||||
Ok(RearchiveResult {
|
Ok(RearchiveResult {
|
||||||
status: "completed".to_string(),
|
status: "completed".to_string(),
|
||||||
|
|
@ -2012,7 +2165,10 @@ mod tests {
|
||||||
);
|
);
|
||||||
// Full YouTube URLs pass through unchanged
|
// Full YouTube URLs pass through unchanged
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
expand_shorthand_to_url("https://www.youtube.com/watch?v=UHxw-L2WyyY", &Source::YouTubeVideo),
|
expand_shorthand_to_url(
|
||||||
|
"https://www.youtube.com/watch?v=UHxw-L2WyyY",
|
||||||
|
&Source::YouTubeVideo
|
||||||
|
),
|
||||||
"https://www.youtube.com/watch?v=UHxw-L2WyyY"
|
"https://www.youtube.com/watch?v=UHxw-L2WyyY"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
@ -2142,14 +2298,32 @@ mod tests {
|
||||||
expected: Source::YouTubeChannel,
|
expected: Source::YouTubeChannel,
|
||||||
},
|
},
|
||||||
// Bare video ID — exactly 11 chars [A-Za-z0-9_-]
|
// Bare video ID — exactly 11 chars [A-Za-z0-9_-]
|
||||||
TestCase { url: "yt:dQw4w9WgXcQ", expected: Source::YouTubeVideo },
|
TestCase {
|
||||||
TestCase { url: "youtube:dQw4w9WgXcQ", expected: Source::YouTubeVideo },
|
url: "yt:dQw4w9WgXcQ",
|
||||||
TestCase { url: "yt:a_b-c_d-e_4", expected: Source::YouTubeVideo },
|
expected: Source::YouTubeVideo,
|
||||||
|
},
|
||||||
|
TestCase {
|
||||||
|
url: "youtube:dQw4w9WgXcQ",
|
||||||
|
expected: Source::YouTubeVideo,
|
||||||
|
},
|
||||||
|
TestCase {
|
||||||
|
url: "yt:a_b-c_d-e_4",
|
||||||
|
expected: Source::YouTubeVideo,
|
||||||
|
},
|
||||||
// Non-ID: wrong length (9 chars) → Other
|
// Non-ID: wrong length (9 chars) → Other
|
||||||
TestCase { url: "yt:not-video", expected: Source::Other },
|
TestCase {
|
||||||
|
url: "yt:not-video",
|
||||||
|
expected: Source::Other,
|
||||||
|
},
|
||||||
// Reserved prefixes still route correctly when segment looks like an ID
|
// Reserved prefixes still route correctly when segment looks like an ID
|
||||||
TestCase { url: "yt:playlist/dQw4w9WgXcQ", expected: Source::YouTubePlaylist },
|
TestCase {
|
||||||
TestCase { url: "yt:@dQw4w9WgXcQ", expected: Source::YouTubeChannel },
|
url: "yt:playlist/dQw4w9WgXcQ",
|
||||||
|
expected: Source::YouTubePlaylist,
|
||||||
|
},
|
||||||
|
TestCase {
|
||||||
|
url: "yt:@dQw4w9WgXcQ",
|
||||||
|
expected: Source::YouTubeChannel,
|
||||||
|
},
|
||||||
];
|
];
|
||||||
|
|
||||||
for case in &shorthand_cases {
|
for case in &shorthand_cases {
|
||||||
|
|
@ -2165,11 +2339,17 @@ mod tests {
|
||||||
#[test]
|
#[test]
|
||||||
fn test_is_youtube_video_id() {
|
fn test_is_youtube_video_id() {
|
||||||
assert!(is_youtube_video_id("dQw4w9WgXcQ"), "canonical ID");
|
assert!(is_youtube_video_id("dQw4w9WgXcQ"), "canonical ID");
|
||||||
assert!(is_youtube_video_id("a_b-c_d-e_4"), "11-char ID with _ and -");
|
assert!(
|
||||||
|
is_youtube_video_id("a_b-c_d-e_4"),
|
||||||
|
"11-char ID with _ and -"
|
||||||
|
);
|
||||||
assert!(!is_youtube_video_id(""), "empty");
|
assert!(!is_youtube_video_id(""), "empty");
|
||||||
assert!(!is_youtube_video_id("short"), "too short");
|
assert!(!is_youtube_video_id("short"), "too short");
|
||||||
assert!(!is_youtube_video_id("toolong12345"), "too long (12 chars)");
|
assert!(!is_youtube_video_id("toolong12345"), "too long (12 chars)");
|
||||||
assert!(!is_youtube_video_id("dQw4w9WgXc!"), "invalid char (! at pos 11)");
|
assert!(
|
||||||
|
!is_youtube_video_id("dQw4w9WgXc!"),
|
||||||
|
"invalid char (! at pos 11)"
|
||||||
|
);
|
||||||
assert!(!is_youtube_video_id("not-video"), "9 chars — too short");
|
assert!(!is_youtube_video_id("not-video"), "9 chars — too short");
|
||||||
assert!(!is_youtube_video_id("dQw4w9WgXC!!"), "12 chars + bad char");
|
assert!(!is_youtube_video_id("dQw4w9WgXC!!"), "12 chars + bad char");
|
||||||
}
|
}
|
||||||
|
|
@ -2289,11 +2469,17 @@ mod tests {
|
||||||
|
|
||||||
// --- expand_shorthand_to_url ---
|
// --- expand_shorthand_to_url ---
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
expand_shorthand_to_url("spotify:track:4iV5W9uYEdYUVa79Axb7Rh", &Source::SpotifyTrack),
|
expand_shorthand_to_url(
|
||||||
|
"spotify:track:4iV5W9uYEdYUVa79Axb7Rh",
|
||||||
|
&Source::SpotifyTrack
|
||||||
|
),
|
||||||
"https://open.spotify.com/track/4iV5W9uYEdYUVa79Axb7Rh"
|
"https://open.spotify.com/track/4iV5W9uYEdYUVa79Axb7Rh"
|
||||||
);
|
);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
expand_shorthand_to_url("spotify:album:1DFixLWuPkv3KT3TnV35m3", &Source::SpotifyAlbum),
|
expand_shorthand_to_url(
|
||||||
|
"spotify:album:1DFixLWuPkv3KT3TnV35m3",
|
||||||
|
&Source::SpotifyAlbum
|
||||||
|
),
|
||||||
"https://open.spotify.com/album/1DFixLWuPkv3KT3TnV35m3"
|
"https://open.spotify.com/album/1DFixLWuPkv3KT3TnV35m3"
|
||||||
);
|
);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
|
|
@ -2305,9 +2491,18 @@ mod tests {
|
||||||
);
|
);
|
||||||
|
|
||||||
// --- source_metadata ---
|
// --- source_metadata ---
|
||||||
assert_eq!(source_metadata(Source::SpotifyTrack), ("spotify", "music", "audio"));
|
assert_eq!(
|
||||||
assert_eq!(source_metadata(Source::SpotifyAlbum), ("spotify", "album", "container"));
|
source_metadata(Source::SpotifyTrack),
|
||||||
assert_eq!(source_metadata(Source::SpotifyPlaylist), ("spotify", "playlist", "container"));
|
("spotify", "music", "audio")
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
source_metadata(Source::SpotifyAlbum),
|
||||||
|
("spotify", "album", "container")
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
source_metadata(Source::SpotifyPlaylist),
|
||||||
|
("spotify", "playlist", "container")
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|
@ -2566,25 +2761,37 @@ mod tests {
|
||||||
#[test]
|
#[test]
|
||||||
fn youtube_video_uses_title() {
|
fn youtube_video_uses_title() {
|
||||||
let m = meta(None, Some("How to Rust"), None, None, None);
|
let m = meta(None, Some("How to Rust"), None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::YouTubeVideo, &m), "How to Rust");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::YouTubeVideo, &m),
|
||||||
|
"How to Rust"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn youtube_video_fallback() {
|
fn youtube_video_fallback() {
|
||||||
let m = meta(None, None, None, None, None);
|
let m = meta(None, None, None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::YouTubeVideo, &m), "YouTube Video");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::YouTubeVideo, &m),
|
||||||
|
"YouTube Video"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn youtube_playlist_uses_title() {
|
fn youtube_playlist_uses_title() {
|
||||||
let m = meta(None, Some("Rust Tutorial Series"), None, None, None);
|
let m = meta(None, Some("Rust Tutorial Series"), None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::YouTubePlaylist, &m), "Rust Tutorial Series");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::YouTubePlaylist, &m),
|
||||||
|
"Rust Tutorial Series"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn youtube_channel_uses_author() {
|
fn youtube_channel_uses_author() {
|
||||||
let m = meta(Some("Rust By Example"), None, None, None, None);
|
let m = meta(Some("Rust By Example"), None, None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::YouTubeChannel, &m), "Archival of Rust By Example");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::YouTubeChannel, &m),
|
||||||
|
"Archival of Rust By Example"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|
@ -2596,7 +2803,10 @@ mod tests {
|
||||||
#[test]
|
#[test]
|
||||||
fn tweet_uses_excerpt_and_author() {
|
fn tweet_uses_excerpt_and_author() {
|
||||||
let m = meta(Some("alice"), None, Some("Hello world"), None, None);
|
let m = meta(Some("alice"), None, Some("Hello world"), None, None);
|
||||||
assert_eq!(generate_entry_title(Source::Tweet, &m), "Hello world \u{2014} @alice");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::Tweet, &m),
|
||||||
|
"Hello world \u{2014} @alice"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|
@ -2612,30 +2822,48 @@ mod tests {
|
||||||
#[test]
|
#[test]
|
||||||
fn tweet_thread_uses_author() {
|
fn tweet_thread_uses_author() {
|
||||||
let m = meta(Some("bob"), None, None, None, None);
|
let m = meta(Some("bob"), None, None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::TweetThread, &m), "Thread by @bob");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::TweetThread, &m),
|
||||||
|
"Thread by @bob"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn instagram_uses_author() {
|
fn instagram_uses_author() {
|
||||||
let m = meta(Some("photographer"), None, None, None, None);
|
let m = meta(Some("photographer"), None, None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::Instagram, &m), "Post by @photographer");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::Instagram, &m),
|
||||||
|
"Post by @photographer"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn facebook_uses_author_no_at() {
|
fn facebook_uses_author_no_at() {
|
||||||
let m = meta(Some("John Doe"), None, None, None, None);
|
let m = meta(Some("John Doe"), None, None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::Facebook, &m), "Post by John Doe");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::Facebook, &m),
|
||||||
|
"Post by John Doe"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn tiktok_uses_author() {
|
fn tiktok_uses_author() {
|
||||||
let m = meta(Some("dancemaster"), None, None, None, None);
|
let m = meta(Some("dancemaster"), None, None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::TikTok, &m), "TikTok by @dancemaster");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::TikTok, &m),
|
||||||
|
"TikTok by @dancemaster"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn reddit_full_fields() {
|
fn reddit_full_fields() {
|
||||||
let m = meta(None, Some("My first Rust project"), None, Some("rust"), Some("newbie"));
|
let m = meta(
|
||||||
|
None,
|
||||||
|
Some("My first Rust project"),
|
||||||
|
None,
|
||||||
|
Some("rust"),
|
||||||
|
Some("newbie"),
|
||||||
|
);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
generate_entry_title(Source::Reddit, &m),
|
generate_entry_title(Source::Reddit, &m),
|
||||||
"My first Rust project \u{2014} r/rust (u/newbie)"
|
"My first Rust project \u{2014} r/rust (u/newbie)"
|
||||||
|
|
@ -2645,7 +2873,10 @@ mod tests {
|
||||||
#[test]
|
#[test]
|
||||||
fn snapchat_uses_author() {
|
fn snapchat_uses_author() {
|
||||||
let m = meta(Some("snapuser"), None, None, None, None);
|
let m = meta(Some("snapuser"), None, None, None, None);
|
||||||
assert_eq!(generate_entry_title(Source::Snapchat, &m), "Snap by snapuser");
|
assert_eq!(
|
||||||
|
generate_entry_title(Source::Snapchat, &m),
|
||||||
|
"Snap by snapuser"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|
@ -2668,7 +2899,10 @@ mod tests {
|
||||||
}"#;
|
}"#;
|
||||||
let meta = tweet_metadata_from_json(json);
|
let meta = tweet_metadata_from_json(json);
|
||||||
assert_eq!(meta.author, Some("rustacean".to_string()));
|
assert_eq!(meta.author, Some("rustacean".to_string()));
|
||||||
assert_eq!(meta.caption, Some("Hello Rust world, this is a test tweet".to_string()));
|
assert_eq!(
|
||||||
|
meta.caption,
|
||||||
|
Some("Hello Rust world, this is a test tweet".to_string())
|
||||||
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -2735,4 +2969,94 @@ mod tests {
|
||||||
assert_eq!(locator_to_playlist_url("https://example.com/page"), None);
|
assert_eq!(locator_to_playlist_url("https://example.com/page"), None);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
mod freedium_supported_url_tests {
|
||||||
|
use super::is_freedium_supported_url;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn medium_bare_domain() {
|
||||||
|
assert!(is_freedium_supported_url("https://medium.com/some/article"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn medium_subdomain() {
|
||||||
|
// Custom Medium publication domains use *.medium.com
|
||||||
|
assert!(is_freedium_supported_url(
|
||||||
|
"https://towardsdatascience.medium.com/article-slug-abc123"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn nytimes() {
|
||||||
|
assert!(is_freedium_supported_url(
|
||||||
|
"https://www.nytimes.com/2024/01/01/tech/ai.html"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn washingtonpost() {
|
||||||
|
assert!(is_freedium_supported_url(
|
||||||
|
"https://www.washingtonpost.com/technology/article"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bloomberg() {
|
||||||
|
assert!(is_freedium_supported_url(
|
||||||
|
"https://bloomberg.com/news/articles/2024-01-01/story"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn reuters() {
|
||||||
|
assert!(is_freedium_supported_url(
|
||||||
|
"https://www.reuters.com/technology/story-2024"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn economist() {
|
||||||
|
assert!(is_freedium_supported_url(
|
||||||
|
"https://www.economist.com/science-and-technology/2024/01/01/article"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn financial_times() {
|
||||||
|
assert!(is_freedium_supported_url(
|
||||||
|
"https://www.ft.com/content/some-uuid"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn non_paywall_site_rejected() {
|
||||||
|
// The original bug: borretti.me was incorrectly routed through Freedium
|
||||||
|
assert!(!is_freedium_supported_url(
|
||||||
|
"https://borretti.me/article/notes-on-managing-adhd"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn generic_url_rejected() {
|
||||||
|
assert!(!is_freedium_supported_url("https://example.com/page"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn freedium_mirror_itself_rejected() {
|
||||||
|
// Existing check already blocks mirror re-wrapping, but belt-and-suspenders
|
||||||
|
assert!(!is_freedium_supported_url(
|
||||||
|
"https://freedium-mirror.cfd/https://medium.com/article"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unparseable_url_rejected() {
|
||||||
|
assert!(!is_freedium_supported_url("not a url at all"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn lookalike_subdomain_rejected() {
|
||||||
|
// "notmedium.com" should not match due to the dot-prefix check
|
||||||
|
assert!(!is_freedium_supported_url("https://notmedium.com/article"));
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue