1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-10-09 21:03:17 +02:00

Add YouTube subtitles, local transcription, self-updating yt-dlp/Deno, X Article and thread titles

- Capture YouTube subtitles by default (opt-out in UI, API, CLI --no-subtitles)
- Summarize YouTube videos from subtitles; fetch on demand, then local transcription, then error
- Local transcription fallback: Whisper, Parakeet, Phonon-2 (English only)
- Runtime-resolved, self-updating yt-dlp and Deno JS runtime (fixes YouTube 403s)
- Settings > Instance > yt-dlp: status and in-app update without restart
- X Article titles from article.title, with idempotent startup backfill
- Thread title generation (single and bulk) with per-provider cheap models
- Per-provider title model settings in Settings > Instance
- Docs, mental model, AGENTS.md and transcription spec updated
This commit is contained in:
TheGeneralist 2026-10-05 00:30:39 +02:00
parent 4f3b2968b6
commit 253f779216
35 changed files with 11377 additions and 583 deletions

View file

@ -15,6 +15,10 @@ sha3.workspace = true
uuid.workspace = true
reqwest = { workspace = true }
base64.workspace = true
zip.workspace = true
[target.'cfg(unix)'.dependencies]
libc.workspace = true
[dev-dependencies]
tempfile = "3"

View file

@ -1,6 +1,6 @@
use crate::{
archive::{self, ArchivePaths},
database, downloader,
database, downloader, subtitles,
twitter::parse_tweet_id,
};
use anyhow::{Context, Result};
@ -83,7 +83,7 @@ impl PlatformMetadata {
/// Configuration passed to `perform_capture` to supply per-instance settings
/// that live outside the archive (e.g. cookies stored in the auth DB).
#[derive(Debug, Clone, Default)]
#[derive(Debug, Clone)]
pub struct CaptureConfig {
pub cookie_rules: Vec<database::CookieRule>,
/// Override for uBlock Origin Lite during WebPage captures.
@ -108,6 +108,38 @@ pub struct CaptureConfig {
/// When true, skip playlist items whose URL is already archived as a child
/// of any container entry with the same canonical playlist URL.
pub sync: bool,
/// Download subtitles for YouTube videos (manual preferred, auto fallback). Default true.
pub download_subtitles: bool,
}
impl Default for CaptureConfig {
fn default() -> Self {
Self {
cookie_rules: Vec::new(),
ublock_enabled: None,
cookie_ext_enabled: None,
reader_mode: false,
modal_closer_enabled: None,
via_freedium: false,
per_item_quality: HashMap::new(),
sync: false,
download_subtitles: true,
}
}
}
/// Plan a subtitle request from the yt-dlp metadata probe. Only YouTube videos
/// (not YouTube Music / audio) get subtitles, and only when enabled.
fn subtitle_request_for(
source: Source,
config: &CaptureConfig,
metadata_json: Option<&str>,
) -> Option<downloader::ytdlp::SubtitleRequest> {
if config.download_subtitles && source == Source::YouTubeVideo {
downloader::ytdlp::plan_subtitle_request(metadata_json)
} else {
None
}
}
/// Resolves which cookies apply to `url` by evaluating all rules in ordinal order.
@ -238,12 +270,17 @@ fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
.unwrap_or_else(|| "Spotify Content".to_string()),
Source::X => format!("X Media by {}", meta.author.as_deref().unwrap_or("unknown")),
Source::Tweet => {
let excerpt = meta
.caption_excerpt()
let headline = meta
.title
.as_deref()
.map(str::trim)
.filter(|t| !t.is_empty())
.map(str::to_string)
.or_else(|| meta.caption_excerpt())
.unwrap_or_else(|| "Tweet".to_string());
format!(
"{} \u{2014} @{}",
excerpt,
headline,
meta.author.as_deref().unwrap_or("unknown")
)
}
@ -911,6 +948,7 @@ fn record_container_entry(
}
/// Extracts PlatformMetadata from a tweet JSON string.
/// `title` is the X Article title when the status is an Article.
/// Returns Default on any parse failure.
fn tweet_metadata_from_json(json_str: &str) -> PlatformMetadata {
let Ok(v) = serde_json::from_str::<serde_json::Value>(json_str) else {
@ -930,8 +968,16 @@ fn tweet_metadata_from_json(json_str: &str) -> PlatformMetadata {
.map(|s| s.trim().to_string())
.filter(|s| !s.is_empty());
let article_title = v
.get("article")
.and_then(|a| a.get("title"))
.and_then(|t| t.as_str())
.map(|s| s.trim().to_string())
.filter(|s| !s.is_empty());
PlatformMetadata {
author: screen_name,
title: article_title,
caption: full_text,
..Default::default()
}
@ -1050,6 +1096,69 @@ fn record_tweet_entry(
Ok(entry)
}
/// Rewrites legacy bare-link titles of X Article tweet entries to
/// "<article title> — @handle". Idempotent and safe to run on every start:
/// a row is only rewritten while its title still byte-equals the legacy title
/// recomputed from its raw JSON (so user renames are never touched), and the
/// write is compare-and-set. Rows with missing/unreadable raw JSON are skipped
/// with a `warn:` line and retried next run. Returns the number of rows changed.
pub fn backfill_x_article_titles(paths: &archive::ArchivePaths) -> Result<usize> {
let conn = database::open_or_initialize(&paths.archive_path)?;
let mut count = 0;
for c in database::list_bare_link_tweet_titles(&conn)? {
let Ok(source_meta) = serde_json::from_str::<serde_json::Value>(&c.source_metadata_json)
else {
continue;
};
let Some(tweet_id) = source_meta.get("tweet_id").and_then(|v| v.as_str()) else {
continue;
};
if tweet_id.is_empty() || !tweet_id.bytes().all(|b| b.is_ascii_digit()) {
continue;
}
let json_path = paths
.store_path
.join("raw_tweets")
.join(format!("tweet-{tweet_id}.json"));
let json = match fs::read_to_string(&json_path) {
Ok(json) => json,
Err(e) => {
eprintln!("warn: X Article title backfill: skipping entry {}: {e}", c.id);
continue;
}
};
let meta = tweet_metadata_from_json(&json);
if meta.title.is_none() {
continue;
}
let bare_link = meta.caption.as_deref().map(str::trim).is_some_and(|t| {
!t.chars().any(char::is_whitespace)
&& (t.starts_with("https://") || t.starts_with("http://"))
});
if !bare_link {
continue;
}
let legacy = generate_entry_title(
Source::Tweet,
&PlatformMetadata {
title: None,
..meta.clone()
},
);
if c.title != legacy {
continue;
}
let new_title = generate_entry_title(Source::Tweet, &meta);
if new_title == legacy {
continue;
}
if database::replace_entry_title_if_unchanged(&conn, c.id, &legacy, &new_title)? {
count += 1;
}
}
Ok(count)
}
/// Trusted image MIME types emitted by the X downloader's media paths.
///
/// Tweet JSON has no MIME field for ordinary downloaded media. Restricting this
@ -1375,15 +1484,26 @@ pub fn perform_capture(
.or(child_quality)
};
let child_source = match source {
Source::SpotifyAlbum | Source::SpotifyPlaylist => Source::SpotifyTrack,
_ if is_audio => Source::YouTubeMusicTrack,
_ => Source::YouTubeVideo,
};
let child_subtitle_request =
subtitle_request_for(child_source, config, child_meta_json.as_deref());
// Download the media.
match downloader::ytdlp::download(
playlist_item.url.clone(),
store_path,
&child_timestamp,
effective_child_quality,
child_subtitle_request.as_ref(),
&cookies,
) {
Ok((hash, file_extension)) => {
Ok(dl) => {
let hash = dl.hash;
let file_extension = dl.extension;
let temp_file = store_path
.join("temp")
.join(&child_timestamp)
@ -1411,13 +1531,10 @@ pub fn perform_capture(
continue;
}
}
let archived_subtitles =
subtitles::archive_staged_subtitles(store_path, dl.subtitles);
let _ = fs::remove_dir_all(store_path.join("temp").join(&child_timestamp));
let child_source = match source {
Source::SpotifyAlbum | Source::SpotifyPlaylist => Source::SpotifyTrack,
_ if is_audio => Source::YouTubeMusicTrack,
_ => Source::YouTubeVideo,
};
match record_media_entry(
&conn,
store_path,
@ -1435,6 +1552,18 @@ pub fn perform_capture(
Some(container_id),
) {
Ok(child_entry) => {
if let Err(e) = subtitles::register_subtitle_artifacts(
&conn,
store_path,
child_entry.id,
&archived_subtitles,
subtitles::SUBTITLE_ORIGIN_CAPTURE,
) {
eprintln!(
"warn: register subtitles for {}: {e:#}",
playlist_item.url
);
}
let _ = database::refresh_entry_cached_bytes(&conn, child_entry.id);
}
Err(e) => {
@ -1829,7 +1958,9 @@ pub fn perform_capture(
_ => None,
};
let (hash, file_extension) = match source {
let subtitle_request = subtitle_request_for(source, config, ytdlp_metadata_json.as_deref());
let (hash, file_extension, staged_subtitles) = match source {
Source::YouTubeVideo
| Source::X
| Source::Instagram
@ -1842,10 +1973,12 @@ pub fn perform_capture(
store_path,
&timestamp,
quality,
subtitle_request.as_ref(),
&cookies,
) {
Ok(result) => result,
Ok(d) => (d.hash, d.extension, d.subtitles),
Err(e) => {
let _ = fs::remove_dir_all(store_path.join("temp").join(&timestamp));
return Err(fail_run(
&conn,
&run,
@ -1862,10 +1995,12 @@ pub fn perform_capture(
store_path,
&timestamp,
Some("audio"),
None,
&cookies,
) {
Ok(result) => result,
Ok(d) => (d.hash, d.extension, d.subtitles),
Err(e) => {
let _ = fs::remove_dir_all(store_path.join("temp").join(&timestamp));
return Err(fail_run(
&conn,
&run,
@ -1876,7 +2011,7 @@ pub fn perform_capture(
}
}
Source::Local => match downloader::local::save(path.clone(), store_path, &timestamp) {
Ok(h) => (h, local_file_extension(&path)),
Ok(h) => (h, local_file_extension(&path), Vec::new()),
Err(e) => {
return Err(fail_run(
&conn,
@ -1893,9 +2028,17 @@ pub fn perform_capture(
.join("temp")
.join(&timestamp)
.join(format!("{timestamp}{file_extension}"));
let byte_size = fs::metadata(&temp_file)
.with_context(|| format!("failed to stat staged file {}", temp_file.display()))?
.len() as i64;
let byte_size = match fs::metadata(&temp_file) {
Ok(meta) => meta.len() as i64,
Err(e) => {
let _ = fs::remove_dir_all(store_path.join("temp").join(&timestamp));
return Err(anyhow::Error::new(e)
.context(format!("failed to stat staged file {}", temp_file.display())));
}
};
// Archive subtitle sidecars before the temp dir is removed below.
let archived_subtitles = subtitles::archive_staged_subtitles(store_path, staged_subtitles);
let hash_exists = hash_exists(&hash, &file_extension, store_path)?;
@ -1942,6 +2085,15 @@ pub fn perform_capture(
None,
None,
)?;
if let Err(e) = subtitles::register_subtitle_artifacts(
&conn,
store_path,
media_entry.id,
&archived_subtitles,
subtitles::SUBTITLE_ORIGIN_CAPTURE,
) {
eprintln!("warn: register subtitles for {path}: {e:#}");
}
database::refresh_entry_cached_bytes(&conn, media_entry.id)?;
database::finish_archive_run(&conn, run.id)?;
@ -3452,6 +3604,163 @@ mod tests {
meta.caption,
Some("Hello Rust world, this is a test tweet".to_string())
);
assert_eq!(meta.title, None);
}
#[test]
fn tweet_prefers_article_title() {
let m = meta(
Some("undefinedKi"),
Some("Why Boring Wins"),
Some("https://t.co/sDrzjUhCzy"),
None,
None,
);
assert_eq!(
generate_entry_title(Source::Tweet, &m),
"Why Boring Wins \u{2014} @undefinedKi"
);
}
#[test]
fn tweet_blank_article_title_falls_back_to_caption() {
let m = meta(Some("alice"), Some(" "), Some("Hello"), None, None);
assert_eq!(
generate_entry_title(Source::Tweet, &m),
"Hello \u{2014} @alice"
);
}
#[test]
fn tweet_title_extracted_from_x_article_json() {
let json = r#"{"full_text":"https://t.co/sDrzjUhCzy","author":{"screen_name":"undefinedKi"},"is_article":true,"article":{"title":" Why Boring Wins ","plain_text":"body"}}"#;
let meta = tweet_metadata_from_json(json);
assert_eq!(meta.title.as_deref(), Some("Why Boring Wins"));
assert_eq!(meta.caption.as_deref(), Some("https://t.co/sDrzjUhCzy"));
assert_eq!(
generate_entry_title(Source::Tweet, &meta),
"Why Boring Wins \u{2014} @undefinedKi"
);
}
}
mod x_article_backfill_tests {
use super::*;
const ARTICLE_JSON: &str = r#"{"full_text":"https://t.co/sDrzjUhCzy","author":{"screen_name":"undefinedKi"},"is_article":true,"article":{"title":"Why Boring Wins","plain_text":"body"}}"#;
const PLAIN_LINK_JSON: &str =
r#"{"full_text":"https://t.co/sDrzjUhCzy","author":{"screen_name":"undefinedKi"}}"#;
const LEGACY: &str = "https://t.co/sDrzjUhCzy \u{2014} @undefinedKi";
struct Fixture {
_temp: tempfile::TempDir,
paths: archive::ArchivePaths,
conn: rusqlite::Connection,
entry: database::ArchivedEntry,
}
fn fixture(tweet_json: &str) -> Fixture {
let temp = tempfile::tempdir().unwrap();
let paths = archive::initialize_archive(
temp.path(),
&temp.path().join("store"),
"X Article backfill test",
false,
)
.unwrap();
fs::write(
paths.store_path.join("raw_tweets").join("tweet-555.json"),
tweet_json,
)
.unwrap();
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
let user_id = database::ensure_default_user(&conn).unwrap();
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
let item = database::create_archive_run_item(
&conn, run.id, None, 0, "tweet:555", None, "x", "tweet",
)
.unwrap();
let entry = record_tweet_entry(
&conn,
&paths.store_path,
user_id,
&run,
&item,
"tweet:555",
Source::Tweet,
"555",
&["raw_tweets/tweet-555.json".to_string()],
)
.unwrap();
Fixture {
_temp: temp,
paths,
conn,
entry,
}
}
impl Fixture {
fn set_title(&self, title: &str) {
database::update_entry_title(&self.conn, &self.entry.entry_uid, Some(title))
.unwrap();
}
fn title(&self) -> String {
self.conn
.query_row(
"SELECT title FROM archived_entries WHERE id = ?1",
[self.entry.id],
|row| row.get(0),
)
.unwrap()
}
}
#[test]
fn new_capture_of_article_uses_article_title() {
let f = fixture(ARTICLE_JSON);
assert_eq!(f.title(), "Why Boring Wins \u{2014} @undefinedKi");
}
#[test]
fn backfill_retitles_legacy_bare_link_article_and_is_idempotent() {
let f = fixture(ARTICLE_JSON);
f.set_title(LEGACY);
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 1);
assert_eq!(f.title(), "Why Boring Wins \u{2014} @undefinedKi");
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
assert_eq!(f.title(), "Why Boring Wins \u{2014} @undefinedKi");
}
#[test]
fn backfill_preserves_user_edited_title() {
let f = fixture(ARTICLE_JSON);
f.set_title("My notes");
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
assert_eq!(f.title(), "My notes");
f.set_title("https://t.co/sDrzjUhCzy \u{2014} my pick");
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
assert_eq!(f.title(), "https://t.co/sDrzjUhCzy \u{2014} my pick");
}
#[test]
fn backfill_ignores_bare_link_tweet_without_article() {
let f = fixture(PLAIN_LINK_JSON);
assert_eq!(f.title(), LEGACY);
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
assert_eq!(f.title(), LEGACY);
}
#[test]
fn backfill_skips_missing_raw_json() {
let f = fixture(ARTICLE_JSON);
fs::remove_file(f.paths.store_path.join("raw_tweets").join("tweet-555.json"))
.unwrap();
f.set_title(LEGACY);
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
assert_eq!(f.title(), LEGACY);
}
}
@ -3608,4 +3917,36 @@ mod tests {
assert!(!is_freedium_supported_url("https://notmedium.com/article"));
}
}
#[test]
fn capture_config_default_downloads_subtitles() {
let config = CaptureConfig::default();
assert!(config.download_subtitles);
assert!(config.cookie_rules.is_empty());
assert!(!config.reader_mode);
assert!(!config.via_freedium);
assert!(!config.sync);
assert!(config.per_item_quality.is_empty());
assert_eq!(config.ublock_enabled, None);
assert_eq!(config.cookie_ext_enabled, None);
assert_eq!(config.modal_closer_enabled, None);
}
#[test]
fn subtitle_request_only_for_youtube_video_when_enabled() {
let meta = r#"{"language":"en","subtitles":{"en":[{"ext":"vtt"}]},"automatic_captions":{}}"#;
let enabled = CaptureConfig::default();
let disabled = CaptureConfig {
download_subtitles: false,
..CaptureConfig::default()
};
assert!(subtitle_request_for(Source::YouTubeVideo, &enabled, Some(meta)).is_some());
assert!(subtitle_request_for(Source::YouTubeVideo, &disabled, Some(meta)).is_none());
for source in [Source::TikTok, Source::YouTubeMusicTrack, Source::X] {
assert!(
subtitle_request_for(source, &enabled, Some(meta)).is_none(),
"{source:?} must not request subtitles"
);
}
}
}

View file

@ -176,12 +176,47 @@ pub struct InstanceSettings {
/// A caller may reorder iff `role_bits & reorder_children_role_bits != 0`.
/// Only the Owner may change it. Never contains the Guest bit.
pub reorder_children_role_bits: u32,
/// Admin overrides for the thread-title model per summary provider kind.
/// `None` = fall back to `ARCHIVR_*_TITLE_MODEL` env, then built-in default.
pub title_model_anthropic_http: Option<String>,
pub title_model_openai_compatible: Option<String>,
pub title_model_claude_cli: Option<String>,
pub title_model_codex_cli: Option<String>,
}
impl InstanceSettings {
pub fn can_reorder_children(&self, role_bits: u32) -> bool {
role_bits & self.reorder_children_role_bits != 0
}
/// Instance title-model override for a provider kind (trimmed, non-empty).
pub fn title_model_override(&self, kind: &str) -> Option<&str> {
self.title_model_slot(kind)?
.as_deref()
.map(str::trim)
.filter(|m| !m.is_empty())
}
/// Mutable column for a provider kind's title model; `None` for unknown kinds.
pub fn title_model_slot_mut(&mut self, kind: &str) -> Option<&mut Option<String>> {
match kind {
"anthropic_http" => Some(&mut self.title_model_anthropic_http),
"openai_compatible" => Some(&mut self.title_model_openai_compatible),
"claude_cli" => Some(&mut self.title_model_claude_cli),
"codex_cli" => Some(&mut self.title_model_codex_cli),
_ => None,
}
}
fn title_model_slot(&self, kind: &str) -> Option<&Option<String>> {
match kind {
"anthropic_http" => Some(&self.title_model_anthropic_http),
"openai_compatible" => Some(&self.title_model_openai_compatible),
"claude_cli" => Some(&self.title_model_claude_cli),
"codex_cli" => Some(&self.title_model_codex_cli),
_ => None,
}
}
}
#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)]
@ -666,7 +701,11 @@ pub fn initialize_auth_schema(conn: &Connection) -> Result<()> {
ublock_enabled INTEGER NOT NULL DEFAULT 1 CHECK (ublock_enabled IN (0, 1)),
cookie_ext_enabled INTEGER NOT NULL DEFAULT 1 CHECK (cookie_ext_enabled IN (0, 1)),
modal_closer_enabled INTEGER NOT NULL DEFAULT 1 CHECK (modal_closer_enabled IN (0, 1)),
reorder_children_role_bits INTEGER NOT NULL DEFAULT 12
reorder_children_role_bits INTEGER NOT NULL DEFAULT 12,
title_model_anthropic_http TEXT,
title_model_openai_compatible TEXT,
title_model_claude_cli TEXT,
title_model_codex_cli TEXT
);
INSERT OR IGNORE INTO instance_settings
@ -726,6 +765,18 @@ pub fn initialize_auth_schema(conn: &Connection) -> Result<()> {
"ALTER TABLE instance_settings ADD COLUMN reorder_children_role_bits INTEGER NOT NULL DEFAULT 12",
[],
);
// Add nullable per-provider thread-title model overrides (idempotent migration)
for column in [
"title_model_anthropic_http",
"title_model_openai_compatible",
"title_model_claude_cli",
"title_model_codex_cli",
] {
let _ = conn.execute(
&format!("ALTER TABLE instance_settings ADD COLUMN {column} TEXT"),
[],
);
}
Ok(())
}
@ -952,7 +1003,9 @@ pub fn get_instance_settings(conn: &Connection) -> Result<InstanceSettings> {
COALESCE(ublock_enabled, 1),
COALESCE(cookie_ext_enabled, 1),
COALESCE(modal_closer_enabled, 1),
COALESCE(reorder_children_role_bits, 12)
COALESCE(reorder_children_role_bits, 12),
title_model_anthropic_http, title_model_openai_compatible,
title_model_claude_cli, title_model_codex_cli
FROM instance_settings WHERE id = 1",
[],
|row| {
@ -965,6 +1018,10 @@ pub fn get_instance_settings(conn: &Connection) -> Result<InstanceSettings> {
cookie_ext_enabled: row.get::<_, i64>(5)? != 0,
modal_closer_enabled: row.get::<_, i64>(6)? != 0,
reorder_children_role_bits: row.get::<_, i64>(7)? as u32,
title_model_anthropic_http: row.get(8)?,
title_model_openai_compatible: row.get(9)?,
title_model_claude_cli: row.get(10)?,
title_model_codex_cli: row.get(11)?,
})
},
)
@ -981,7 +1038,11 @@ pub fn update_instance_settings(conn: &Connection, settings: &InstanceSettings)
ublock_enabled = ?5,
cookie_ext_enabled = ?6,
modal_closer_enabled = ?7,
reorder_children_role_bits = ?8
reorder_children_role_bits = ?8,
title_model_anthropic_http = ?9,
title_model_openai_compatible = ?10,
title_model_claude_cli = ?11,
title_model_codex_cli = ?12
WHERE id = 1",
params![
settings.public_index_enabled as i64,
@ -992,6 +1053,10 @@ pub fn update_instance_settings(conn: &Connection, settings: &InstanceSettings)
settings.cookie_ext_enabled as i64,
settings.modal_closer_enabled as i64,
settings.reorder_children_role_bits as i64,
settings.title_model_anthropic_http,
settings.title_model_openai_compatible,
settings.title_model_claude_cli,
settings.title_model_codex_cli,
],
)?;
Ok(())
@ -1121,6 +1186,46 @@ pub fn update_entry_title(conn: &Connection, entry_uid: &str, title: Option<&str
Ok(n > 0)
}
/// An `x`/`tweet` entry whose title may be a legacy bare-link auto title.
#[derive(Debug, Clone)]
pub struct TweetTitleCandidate {
pub id: i64,
pub title: String,
pub source_metadata_json: String,
}
/// Root or child `x`/`tweet` entries whose title starts with "http" (bare-link auto titles).
pub fn list_bare_link_tweet_titles(conn: &Connection) -> Result<Vec<TweetTitleCandidate>> {
let mut stmt = conn.prepare(
"SELECT id, title, source_metadata_json FROM archived_entries
WHERE source_kind = 'x' AND entity_kind = 'tweet' AND title LIKE 'http%'
ORDER BY id",
)?;
let rows = stmt.query_map([], |row| {
Ok(TweetTitleCandidate {
id: row.get(0)?,
title: row.get(1)?,
source_metadata_json: row.get(2)?,
})
})?;
Ok(rows.collect::<rusqlite::Result<Vec<_>>>()?)
}
/// Compare-and-set title update: only writes when the stored title still equals
/// `expected`, so a concurrent rename always wins. `Ok(true)` iff one row changed.
pub fn replace_entry_title_if_unchanged(
conn: &Connection,
entry_id: i64,
expected: &str,
new_title: &str,
) -> Result<bool> {
let n = conn.execute(
"UPDATE archived_entries SET title = ?1 WHERE id = ?2 AND title = ?3",
params![new_title, entry_id, expected],
)?;
Ok(n == 1)
}
/// Outcome of [`reorder_child_entries`]; the server maps it to 204/404/400.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum ReorderChildrenOutcome {
@ -1667,6 +1772,36 @@ pub fn entry_id_for_uid(conn: &Connection, entry_uid: &str) -> Result<Option<i64
.map_err(Into::into)
}
/// An entry's id plus the source identity fields needed to re-fetch it.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct EntrySourceInfo {
pub entry_id: i64,
pub source_kind: String,
pub entity_kind: String,
pub canonical_url: Option<String>,
}
/// Looks up `entry_uid` with its source identity's canonical URL. `Ok(None)` if absent.
pub fn entry_source_info(conn: &Connection, entry_uid: &str) -> Result<Option<EntrySourceInfo>> {
conn.query_row(
"SELECT e.id, e.source_kind, e.entity_kind, si.canonical_url
FROM archived_entries e
JOIN source_identities si ON si.id = e.source_identity_id
WHERE e.entry_uid = ?1",
[entry_uid],
|row| {
Ok(EntrySourceInfo {
entry_id: row.get(0)?,
source_kind: row.get(1)?,
entity_kind: row.get(2)?,
canonical_url: row.get(3)?,
})
},
)
.optional()
.map_err(Into::into)
}
/// Creates a fresh pending summary attempt for one cache key.
///
/// Attempts are intentionally not unique by cache key: a forced regeneration
@ -1751,6 +1886,20 @@ pub fn update_entry_summary_status(
Ok(())
}
/// Replaces a summary row's input digest (used once a deferred input is built).
/// Also bumps `updated_at`.
pub fn update_entry_summary_input_sha256(
conn: &Connection,
summary_uid: &str,
input_sha256: &str,
) -> Result<()> {
conn.execute(
"UPDATE entry_summaries SET input_sha256 = ?1, updated_at = ?2 WHERE summary_uid = ?3",
params![input_sha256, now_timestamp(), summary_uid],
)?;
Ok(())
}
/// Returns one summary by its public uid.
pub fn get_entry_summary_by_uid(
conn: &Connection,
@ -2284,6 +2433,19 @@ pub fn has_active_capture_jobs(conn: &Connection) -> Result<bool> {
Ok(n > 0)
}
/// Returns `true` while a summary-time subtitle fetch is in flight (a
/// `pending`/`running` summary row still carrying the placeholder digest).
/// Like a capture, the fetch moves files into `raw/` before writing DB rows.
pub fn has_pending_subtitle_fetches(conn: &Connection) -> Result<bool> {
let n: i64 = conn.query_row(
"SELECT COUNT(*) FROM entry_summaries
WHERE status IN ('pending', 'running') AND input_sha256 = ?1",
[crate::summarizer::SUBTITLE_FETCH_PENDING_INPUT_SHA256],
|row| row.get(0),
)?;
Ok(n > 0)
}
/// Returns `(id, raw_relpath, byte_size)` for every blob row not referenced by any
/// `entry_artifacts.blob_id`. These DB rows are safe to delete regardless of whether
/// a disk file still exists at their `raw_relpath`.
@ -2436,6 +2598,60 @@ pub fn add_entry_artifact(conn: &Connection, artifact: &NewArtifact) -> Result<i
Ok(conn.last_insert_rowid())
}
/// One artifact of a given role, with its blob MIME type when it has a blob.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct RoleArtifact {
pub id: i64,
pub relpath: String,
pub mime_type: Option<String>,
pub metadata_json: Option<String>,
}
/// Lists an entry's artifacts with `role`, in insertion (id) order.
pub fn list_entry_artifacts_by_role(
conn: &Connection,
entry_id: i64,
role: &str,
) -> Result<Vec<RoleArtifact>> {
let mut stmt = conn.prepare(
"SELECT ea.id, ea.relpath, b.mime_type, ea.metadata_json
FROM entry_artifacts ea
LEFT JOIN blobs b ON b.id = ea.blob_id
WHERE ea.entry_id = ?1 AND ea.artifact_role = ?2
ORDER BY ea.id ASC",
)?;
let rows = stmt
.query_map(params![entry_id, role], |row| {
Ok(RoleArtifact {
id: row.get(0)?,
relpath: row.get(1)?,
mime_type: row.get(2)?,
metadata_json: row.get(3)?,
})
})?
.collect::<rusqlite::Result<Vec<_>>>()?;
Ok(rows)
}
/// True if the entry already has an artifact with `role` pointing at `blob_id`.
/// `entry_artifacts` has no uniqueness constraint, so callers dedupe with this.
pub fn entry_has_artifact_blob(
conn: &Connection,
entry_id: i64,
role: &str,
blob_id: i64,
) -> Result<bool> {
let exists: bool = conn.query_row(
"SELECT EXISTS(
SELECT 1 FROM entry_artifacts
WHERE entry_id = ?1 AND artifact_role = ?2 AND blob_id = ?3
)",
params![entry_id, role, blob_id],
|row| row.get(0),
)?;
Ok(exists)
}
pub fn remove_entry_tag_assignment(conn: &Connection, entry_id: i64, tag_id: i64) -> Result<()> {
conn.execute(
"DELETE FROM entry_tag_assignments WHERE entry_id = ?1 AND tag_id = ?2",
@ -3301,6 +3517,26 @@ mod tests {
.unwrap()
}
#[test]
fn replace_entry_title_if_unchanged_is_compare_and_set() {
let conn = conn();
let entry = create_entry_fixture(&conn, "private", None, None);
update_entry_title(&conn, &entry.entry_uid, Some("a")).unwrap();
let title = |conn: &Connection| -> String {
conn.query_row(
"SELECT title FROM archived_entries WHERE id = ?1",
[entry.id],
|row| row.get(0),
)
.unwrap()
};
assert!(!replace_entry_title_if_unchanged(&conn, entry.id, "b", "c").unwrap());
assert_eq!(title(&conn), "a");
assert!(replace_entry_title_if_unchanged(&conn, entry.id, "a", "c").unwrap());
assert_eq!(title(&conn), "c");
}
#[test]
fn schema_defaults_public_settings_to_private() {
let conn = conn();
@ -3990,6 +4226,29 @@ mod tests {
assert!(s.modal_closer_enabled);
}
#[test]
fn instance_settings_title_models_migrate_and_round_trip() {
let conn = Connection::open_in_memory().unwrap();
conn.execute_batch(
"CREATE TABLE instance_settings (id INTEGER PRIMARY KEY CHECK (id = 1), public_index_enabled INTEGER NOT NULL DEFAULT 0, public_entry_content_enabled INTEGER NOT NULL DEFAULT 0, public_archive_submission_enabled INTEGER NOT NULL DEFAULT 0, default_entry_visibility INTEGER NOT NULL DEFAULT 2);
INSERT INTO instance_settings (id) VALUES (1);",
)
.unwrap();
initialize_auth_schema(&conn).unwrap();
initialize_auth_schema(&conn).unwrap();
let mut s = get_instance_settings(&conn).unwrap();
assert_eq!(s.title_model_claude_cli, None);
assert_eq!(s.title_model_override("claude_cli"), None);
*s.title_model_slot_mut("claude_cli").unwrap() = Some("sonnet".into());
s.title_model_codex_cli = Some(" ".into());
update_instance_settings(&conn, &s).unwrap();
let s = get_instance_settings(&conn).unwrap();
assert_eq!(s.title_model_override("claude_cli"), Some("sonnet"));
assert_eq!(s.title_model_override("codex_cli"), None);
assert_eq!(s.title_model_override("gemini"), None);
assert_eq!(s.title_model_anthropic_http, None);
}
#[test]
fn can_reorder_children_intersects_mask() {
let conn = make_auth_conn_for_mgmt();
@ -4928,6 +5187,22 @@ mod tests {
assert!(rec.completed_at.is_none());
}
#[test]
fn has_pending_subtitle_fetches_tracks_placeholder_rows() {
let c = conn();
let entry = create_entry_fixture(&c, "private", None, None);
let placeholder = crate::summarizer::SUBTITLE_FETCH_PENDING_INPUT_SHA256;
upsert_pending_entry_summary(&c, entry.id, "codex_cli", None, "v1", "real").unwrap();
assert!(!has_pending_subtitle_fetches(&c).unwrap());
let uid =
upsert_pending_entry_summary(&c, entry.id, "codex_cli", None, "v1", placeholder).unwrap();
assert!(has_pending_subtitle_fetches(&c).unwrap());
update_entry_summary_status(&c, &uid, "running", None, None).unwrap();
assert!(has_pending_subtitle_fetches(&c).unwrap());
update_entry_summary_status(&c, &uid, "failed", None, Some("x")).unwrap();
assert!(!has_pending_subtitle_fetches(&c).unwrap());
}
#[test]
fn regenerating_a_completed_summary_creates_a_new_attempt_and_preserves_completion() {
let c = conn();
@ -5275,4 +5550,119 @@ mod tests {
assert_eq!(position_of(&c, b.id), Some(1));
assert_eq!(position_of(&c, c3.id), Some(2));
}
fn test_blob(conn: &Connection, sha: &str, mime: &str) -> i64 {
upsert_blob(
conn,
&BlobRecord {
sha256: sha.to_string(),
byte_size: 10,
mime_type: Some(mime.to_string()),
extension: None,
raw_relpath: format!("raw/{sha}"),
},
)
.unwrap()
}
fn test_artifact(conn: &Connection, entry_id: i64, role: &str, blob_id: Option<i64>, relpath: &str) -> i64 {
add_entry_artifact(
conn,
&NewArtifact {
entry_id,
artifact_role: role.to_string(),
storage_area: "raw".to_string(),
relpath: relpath.to_string(),
blob_id,
logical_path: None,
metadata_json: Some(format!("{{\"r\":\"{relpath}\"}}")),
},
)
.unwrap()
}
#[test]
fn entry_source_info_joins_canonical_url() {
let conn = conn();
let entry = create_entry_fixture(&conn, "private", None, None);
let info = entry_source_info(&conn, &entry.entry_uid).unwrap().unwrap();
assert_eq!(
info,
EntrySourceInfo {
entry_id: entry.id,
source_kind: "youtube".to_string(),
entity_kind: "video".to_string(),
canonical_url: Some("https://youtube.com/watch?v=video-1".to_string()),
}
);
assert!(entry_source_info(&conn, "entry_missing").unwrap().is_none());
}
#[test]
fn list_entry_artifacts_by_role_orders_by_id() {
let conn = conn();
let entry = create_entry_fixture(&conn, "private", None, None);
let vtt = test_blob(&conn, "aa11", "text/vtt");
let first = test_artifact(&conn, entry.id, "subtitle", Some(vtt), "raw/a/a/aa11.vtt");
let _media = test_artifact(&conn, entry.id, "primary_media", None, "raw/m.mp4");
let second = test_artifact(&conn, entry.id, "subtitle", None, "raw/b.srt");
let rows = list_entry_artifacts_by_role(&conn, entry.id, "subtitle").unwrap();
assert_eq!(
rows,
vec![
RoleArtifact {
id: first,
relpath: "raw/a/a/aa11.vtt".to_string(),
mime_type: Some("text/vtt".to_string()),
metadata_json: Some("{\"r\":\"raw/a/a/aa11.vtt\"}".to_string()),
},
RoleArtifact {
id: second,
relpath: "raw/b.srt".to_string(),
mime_type: None,
metadata_json: Some("{\"r\":\"raw/b.srt\"}".to_string()),
},
]
);
assert!(list_entry_artifacts_by_role(&conn, entry.id, "favicon").unwrap().is_empty());
}
#[test]
fn entry_has_artifact_blob_matches_role_and_blob() {
let conn = conn();
let entry = create_entry_fixture(&conn, "private", None, None);
let other = create_entry_fixture(&conn, "private", None, None);
let blob = test_blob(&conn, "bb22", "text/vtt");
let unrelated = test_blob(&conn, "cc33", "text/vtt");
test_artifact(&conn, entry.id, "subtitle", Some(blob), "raw/bb22.vtt");
assert!(entry_has_artifact_blob(&conn, entry.id, "subtitle", blob).unwrap());
assert!(!entry_has_artifact_blob(&conn, entry.id, "primary_media", blob).unwrap());
assert!(!entry_has_artifact_blob(&conn, entry.id, "subtitle", unrelated).unwrap());
assert!(!entry_has_artifact_blob(&conn, other.id, "subtitle", blob).unwrap());
}
#[test]
fn update_entry_summary_input_sha256_updates_row() {
let conn = conn();
let entry = create_entry_fixture(&conn, "private", None, None);
let uid = upsert_pending_entry_summary(
&conn,
entry.id,
"codex_cli",
None,
"v1",
"pending-subtitle-fetch",
)
.unwrap();
let before = get_entry_summary_by_uid(&conn, &uid).unwrap().unwrap();
let digest = "ab".repeat(32);
update_entry_summary_input_sha256(&conn, &uid, &digest).unwrap();
let after = get_entry_summary_by_uid(&conn, &uid).unwrap().unwrap();
assert_eq!(after.input_sha256, digest);
assert_eq!(after.status, "pending");
assert!(after.updated_at >= before.updated_at);
}
}

View file

@ -0,0 +1,328 @@
//! Deno half of the shared yt-dlp tools update (CLI `archivr yt-dlp update` and the admin
//! UI): installs the latest official Deno release into
//! `<state_dir>/deno/deno`, the JS runtime yt-dlp uses to solve YouTube's challenges.
//!
//! The flow mirrors the yt-dlp zipapp install: download, extract into a staging file,
//! verify it runs (`--version` must report exactly the release version), then rename it
//! over the target so a concurrently-running archivr never sees a half-written binary.
//! The release `.sha256sum` is not checked: it comes from the same TLS origin as the zip,
//! and the zip's CRC32 already catches corruption.
use anyhow::{bail, Context, Result};
use super::js_runtime::{
parse_deno_version_output, pinned_deno, probe_deno_version, state_dir_deno, DenoVersion,
MIN_DENO_VERSION,
};
use std::{
env, fs,
io::{self, Cursor},
path::Path,
process::Command,
time::Duration,
};
/// GitHub release metadata endpoint for the upstream Deno project.
pub const DENO_LATEST_RELEASE: &str =
"https://api.github.com/repos/denoland/deno/releases/latest";
/// The Deno zip is ~40 MB; reqwest's blocking client defaults to a 30s total timeout,
/// which is too short on slow links. Applies to the zip download only.
const DENO_DOWNLOAD_TIMEOUT: Duration = Duration::from_secs(600);
/// Why a staged prebuilt Deno can fail to spawn even though the file exists: the official
/// binaries are dynamically linked against a glibc loader NixOS doesn't provide.
const NO_LOADER: &str = "prebuilt deno cannot execute on this host \
(missing dynamic loader — on NixOS enable programs.nix-ld)";
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct DenoRelease {
pub tag: String,
pub version: DenoVersion,
pub download_url: String,
}
/// Official release asset for a `std::env::consts::{OS, ARCH}` pair.
pub fn deno_release_asset(os: &str, arch: &str) -> Option<&'static str> {
match (os, arch) {
("macos", "aarch64") => Some("deno-aarch64-apple-darwin.zip"),
("macos", "x86_64") => Some("deno-x86_64-apple-darwin.zip"),
("linux", "x86_64") => Some("deno-x86_64-unknown-linux-gnu.zip"),
("linux", "aarch64") => Some("deno-aarch64-unknown-linux-gnu.zip"),
_ => None,
}
}
/// Extracts the version and download URL from a GitHub "latest release" response.
/// The URL is built from the tag and asset name rather than taken from the response.
pub fn parse_deno_release(json: &serde_json::Value, asset: &str) -> Result<DenoRelease> {
let tag = json
.get("tag_name")
.and_then(serde_json::Value::as_str)
.context("GitHub releases API response had no tag_name")?;
// The tag ends up in a URL path; only accept plain version-ish characters.
if tag.is_empty()
|| !tag
.chars()
.all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '-' | '+'))
{
bail!("unexpected deno release tag {tag:?}");
}
let version = DenoVersion::parse(tag)
.with_context(|| format!("could not parse a version from deno release tag {tag:?}"))?;
let has_asset = json
.get("assets")
.and_then(serde_json::Value::as_array)
.is_some_and(|assets| {
assets
.iter()
.any(|a| a.get("name").and_then(serde_json::Value::as_str) == Some(asset))
});
if !has_asset {
bail!("deno release {tag} has no {asset} asset");
}
Ok(DenoRelease {
tag: tag.to_string(),
version,
download_url: format!(
"https://github.com/denoland/deno/releases/download/{tag}/{asset}"
),
})
}
/// Installs or updates Deno in the state dir. `Ok` carries a one-line human outcome.
pub fn install_deno(client: &reqwest::blocking::Client, log: &mut dyn FnMut(&str)) -> Result<String> {
let target = state_dir_deno().context("could not determine a state directory (is $HOME set?)")?;
let dir = target
.parent()
.context("state-dir deno path has no parent directory")?;
let staging = dir.join("deno.new");
let (os, arch) = (env::consts::OS, env::consts::ARCH);
let asset = deno_release_asset(os, arch)
.with_context(|| format!("unsupported platform {os}/{arch}"))?;
let body = client
.get(DENO_LATEST_RELEASE)
.send()
.context("failed to reach the GitHub releases API")?
.error_for_status()
.context("GitHub releases API returned an error")?
.text()
.context("failed to read the GitHub releases API response")?;
let json: serde_json::Value =
serde_json::from_str(&body).context("GitHub releases API returned invalid JSON")?;
let release = parse_deno_release(&json, asset)?;
if let Some(installed) = probe_deno_version(&target).filter(|v| *v >= release.version) {
return Ok(format!("deno {installed} already installed at {}", target.display()));
}
log(&format!("Downloading deno {}…", release.version));
let bytes = client
.get(&release.download_url)
.timeout(DENO_DOWNLOAD_TIMEOUT)
.send()
.with_context(|| format!("failed to download {}", release.download_url))?
.error_for_status()
.with_context(|| format!("download of {} failed", release.download_url))?
.bytes()
.context("failed to read the downloaded deno zip")?;
fs::create_dir_all(dir).with_context(|| format!("failed to create {}", dir.display()))?;
let staged = extract_deno(&bytes, &staging)
.and_then(|()| verify_staged(&staging, release.version))
.and_then(|runs| {
if runs {
fs::rename(&staging, &target)
.with_context(|| format!("failed to install {}", target.display()))?;
}
Ok(runs)
});
let runs = staged.inspect_err(|_| {
let _ = fs::remove_file(&staging);
})?;
if !runs {
let _ = fs::remove_file(&staging);
let pinned_usable = pinned_deno()
.and_then(|p| probe_deno_version(&p))
.is_some_and(|v| v >= MIN_DENO_VERSION);
if pinned_usable {
return Ok(format!("skipped: {NO_LOADER}; using pinned ARCHIVR_DENO"));
}
bail!("{NO_LOADER}, and no usable pinned ARCHIVR_DENO is set");
}
Ok(format!("installed deno {} to {}", release.version, target.display()))
}
/// Writes the zip's `deno` entry to `staging` and makes it executable.
fn extract_deno(zip_bytes: &[u8], staging: &Path) -> Result<()> {
let mut archive =
zip::ZipArchive::new(Cursor::new(zip_bytes)).context("downloaded deno zip is invalid")?;
let mut entry = archive
.by_name("deno")
.context("downloaded deno zip has no `deno` entry")?;
let mut out = fs::File::create(staging)
.with_context(|| format!("failed to create {}", staging.display()))?;
io::copy(&mut entry, &mut out)
.with_context(|| format!("failed to extract deno to {}", staging.display()))?;
out.sync_all()
.with_context(|| format!("failed to flush {}", staging.display()))?;
#[cfg(unix)]
{
use std::os::unix::fs::PermissionsExt;
fs::set_permissions(staging, fs::Permissions::from_mode(0o755))
.with_context(|| format!("failed to chmod +x {}", staging.display()))?;
}
Ok(())
}
/// Runs `staging --version` and requires it to report exactly `expected`.
///
/// `Ok(false)` means the binary exists but the OS could not execute it at all
/// (spawn failed with `NotFound` — e.g. the ELF interpreter is missing on NixOS);
/// any other failure is an error.
fn verify_staged(staging: &Path, expected: DenoVersion) -> Result<bool> {
let output = match Command::new(staging).arg("--version").output() {
Ok(output) => output,
Err(e) if e.kind() == io::ErrorKind::NotFound && staging.is_file() => return Ok(false),
Err(e) => {
return Err(e).with_context(|| format!("failed to run {} --version", staging.display()));
}
};
let got = output
.status
.success()
.then(|| parse_deno_version_output(&String::from_utf8_lossy(&output.stdout)))
.flatten();
if got != Some(expected) {
let got = got.map_or_else(|| "no parseable version".to_string(), |v| v.to_string());
bail!("downloaded deno failed verification (expected {expected}, got {got})");
}
Ok(true)
}
#[cfg(test)]
mod tests {
use super::{deno_release_asset, parse_deno_release, verify_staged};
#[cfg(unix)]
use crate::downloader::write_script;
use crate::downloader::js_runtime::DenoVersion;
use serde_json::json;
#[test]
fn supported_platforms_map_to_assets() {
assert_eq!(
deno_release_asset("macos", "aarch64"),
Some("deno-aarch64-apple-darwin.zip")
);
assert_eq!(
deno_release_asset("macos", "x86_64"),
Some("deno-x86_64-apple-darwin.zip")
);
assert_eq!(
deno_release_asset("linux", "x86_64"),
Some("deno-x86_64-unknown-linux-gnu.zip")
);
assert_eq!(
deno_release_asset("linux", "aarch64"),
Some("deno-aarch64-unknown-linux-gnu.zip")
);
}
#[test]
fn unsupported_platforms_have_no_asset() {
assert_eq!(deno_release_asset("windows", "x86_64"), None);
assert_eq!(deno_release_asset("linux", "riscv64"), None);
}
#[test]
fn release_json_yields_version_and_url() {
let json = json!({
"tag_name": "v2.9.7",
"assets": [
{"name": "deno-x86_64-unknown-linux-gnu.zip"},
{"name": "deno-aarch64-apple-darwin.zip"},
],
});
let release = parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").unwrap();
assert_eq!(release.tag, "v2.9.7");
assert_eq!(
release.version,
DenoVersion { major: 2, minor: 9, patch: 7 }
);
assert_eq!(
release.download_url,
"https://github.com/denoland/deno/releases/download/v2.9.7/deno-aarch64-apple-darwin.zip"
);
}
#[test]
fn missing_asset_is_an_error_naming_it() {
let json = json!({
"tag_name": "v2.9.7",
"assets": [{"name": "deno-x86_64-unknown-linux-gnu.zip"}],
});
let err = parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").unwrap_err();
assert!(
err.to_string().contains("deno-aarch64-apple-darwin.zip"),
"{err:#}"
);
}
#[test]
fn missing_or_bad_tag_is_an_error() {
let assets = json!([{"name": "deno-aarch64-apple-darwin.zip"}]);
for json in [
json!({"assets": assets}),
json!({"tag_name": 297, "assets": assets}),
json!({"tag_name": "nightly", "assets": assets}),
json!({"tag_name": "v2.9.7/../../evil", "assets": assets}),
] {
assert!(
parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").is_err(),
"{json}"
);
}
}
#[cfg(unix)]
#[test]
fn staged_binary_must_report_the_release_version() {
let tmp = tempfile::tempdir().unwrap();
let expected = DenoVersion { major: 2, minor: 9, patch: 7 };
// A fresh path per script: rewriting one path that was just exec'd invites ETXTBSY.
let good = tmp.path().join("good/deno.new");
write_script(&good, "#!/bin/sh\necho 'deno 2.9.7 (stable, release, test)'\n");
assert!(verify_staged(&good, expected).unwrap());
let wrong = tmp.path().join("wrong/deno.new");
write_script(&wrong, "#!/bin/sh\necho 'deno 2.9.6 (stable, release, test)'\n");
let err = verify_staged(&wrong, expected).unwrap_err();
assert!(err.to_string().contains("expected 2.9.7, got 2.9.6"), "{err:#}");
let failing = tmp.path().join("failing/deno.new");
write_script(&failing, "#!/bin/sh\nexit 1\n");
assert!(verify_staged(&failing, expected).is_err());
}
#[cfg(unix)]
#[test]
fn existing_but_unexecutable_binary_is_reported_as_cannot_run() {
// A script whose interpreter is missing fails to spawn with NotFound even though
// the file exists — the same shape as a glibc ELF on NixOS without nix-ld.
let tmp = tempfile::tempdir().unwrap();
let staged = tmp.path().join("deno.new");
write_script(&staged, "#!/nonexistent/ld-linux.so\n");
let expected = DenoVersion { major: 2, minor: 9, patch: 7 };
assert!(!verify_staged(&staged, expected).unwrap());
// A genuinely missing file is an error, not "cannot execute".
let missing = tmp.path().join("missing");
assert!(verify_staged(&missing, expected).is_err());
}
}

View file

@ -0,0 +1,684 @@
//! JavaScript runtime resolution for yt-dlp.
//!
//! yt-dlp needs a JS runtime to solve YouTube's challenges (EJS). Without one, YouTube
//! downloads may fail with HTTP 403. This module picks the runtime archivr passes to every
//! yt-dlp process via `--js-runtimes`:
//!
//! 1. `ARCHIVR_JS_RUNTIME` (forced, `RUNTIME[:ABS_PATH]`; skips resolution and version checks).
//! 2. The newest Deno >= [`MIN_DENO_VERSION`] among the pinned `ARCHIVR_DENO` and the
//! state-dir copy (`<state_dir>/deno/deno`); an exact tie goes to the state dir.
//! 3. `deno` on `PATH`, if new enough.
//! 4. Nothing (a warning is printed once per resolution: first use and each refresh).
//!
//! Only Deno is ever chosen automatically; Node, Bun and QuickJS are used only when forced.
use std::{
env,
ffi::{OsStr, OsString},
fmt,
path::{Path, PathBuf},
process::Command,
sync::RwLock,
};
use super::ytdlp::state_dir;
/// Forced runtime override, `RUNTIME[:ABS_PATH]` with RUNTIME one of deno|node|bun|quickjs.
pub const JS_RUNTIME_ENV: &str = "ARCHIVR_JS_RUNTIME";
/// Pinned Deno binary (set by the Nix wrappers and the Docker image).
pub const DENO_ENV: &str = "ARCHIVR_DENO";
/// Oldest Deno yt-dlp's EJS solver supports.
pub const MIN_DENO_VERSION: DenoVersion = DenoVersion { major: 2, minor: 3, patch: 0 };
/// Cached choice: outer `None` = not resolved yet. Swapped by [`refresh_js_runtime`].
static RESOLVED_JS_RUNTIME: RwLock<Option<Option<JsRuntime>>> = RwLock::new(None);
/// Resolves (uncached) and prints the warnings; runs on first use and on each refresh.
fn resolve_js_runtime_logged() -> Option<JsRuntime> {
if let Err(reason) = forced_js_runtime() {
let raw = env::var_os(JS_RUNTIME_ENV).unwrap_or_default();
eprintln!("warn: ignoring {JS_RUNTIME_ENV}={raw:?}: {reason}");
}
let resolved = resolve_js_runtime_with_role().map(|(_, rt)| rt);
if resolved.is_none() {
eprintln!(
"warn: no JavaScript runtime for yt-dlp (need deno >= {MIN_DENO_VERSION} via \
{DENO_ENV}, the state dir, or PATH; or set {JS_RUNTIME_ENV}) — YouTube \
downloads may fail with HTTP 403; run `archivr yt-dlp update` to install deno"
);
}
resolved
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum JsRuntimeKind {
Deno,
Node,
Bun,
QuickJs,
}
impl JsRuntimeKind {
/// Runtime name as yt-dlp's `--js-runtimes` expects it.
pub fn as_str(self) -> &'static str {
match self {
Self::Deno => "deno",
Self::Node => "node",
Self::Bun => "bun",
Self::QuickJs => "quickjs",
}
}
/// ASCII case-insensitive match against the exact allowlist.
pub fn from_name(name: &str) -> Option<Self> {
[Self::Deno, Self::Node, Self::Bun, Self::QuickJs]
.into_iter()
.find(|k| k.as_str().eq_ignore_ascii_case(name))
}
/// Executable name yt-dlp looks for when the runtime path is a directory
/// (mirrors `_determine_runtime_path` in yt-dlp's JS runtime classes).
fn executable_name(self) -> &'static str {
match self {
Self::QuickJs => "qjs",
other => other.as_str(),
}
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct JsRuntime {
pub kind: JsRuntimeKind,
pub path: Option<PathBuf>,
}
impl JsRuntime {
/// `kind` or `kind:path`, built without lossy conversion.
pub fn spec(&self) -> OsString {
let mut spec = OsString::from(self.kind.as_str());
if let Some(path) = &self.path {
spec.push(":");
spec.push(path.as_os_str());
}
spec
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
pub struct DenoVersion {
pub major: u64,
pub minor: u64,
pub patch: u64,
}
impl DenoVersion {
/// Parses `2.9.7` or `v2.9.7`; anything after the patch digits (`+abc`, `-rc1`) is ignored.
pub fn parse(s: &str) -> Option<Self> {
let s = s.trim();
let s = s.strip_prefix('v').unwrap_or(s);
let end = s
.find(|c: char| !(c.is_ascii_digit() || c == '.'))
.unwrap_or(s.len());
let mut parts = s[..end].split('.').map(|p| p.parse::<u64>().ok());
let version = Self {
major: parts.next()??,
minor: parts.next()??,
patch: parts.next()??,
};
parts.next().is_none().then_some(version)
}
}
impl fmt::Display for DenoVersion {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "{}.{}.{}", self.major, self.minor, self.patch)
}
}
/// Validates a `RUNTIME[:ABS_PATH]` spec. The path may be a file or a directory
/// (yt-dlp accepts both) but must be absolute and exist.
pub fn parse_js_runtime_spec(raw: &str) -> Result<JsRuntime, String> {
let raw = raw.trim();
let (name, path) = match raw.split_once(':') {
Some((name, path)) => (name, Some(path)),
None => (raw, None),
};
let kind = JsRuntimeKind::from_name(name).ok_or_else(|| {
format!("unknown runtime {name} (expected deno, node, bun or quickjs)")
})?;
let path = match path {
None => None,
Some("") => return Err("empty path after ':'".to_string()),
Some(p) => {
let p = PathBuf::from(p);
if !p.is_absolute() {
return Err("path must be absolute".to_string());
}
if !p.exists() {
return Err("path does not exist".to_string());
}
Some(p)
}
};
Ok(JsRuntime { kind, path })
}
/// Reads `ARCHIVR_JS_RUNTIME`; `Ok(None)` if unset or empty.
pub fn forced_js_runtime() -> Result<Option<JsRuntime>, String> {
match env::var(JS_RUNTIME_ENV) {
Err(env::VarError::NotPresent) => Ok(None),
Err(env::VarError::NotUnicode(_)) => Err("value is not valid UTF-8".to_string()),
Ok(raw) if raw.trim().is_empty() => Ok(None),
Ok(raw) => parse_js_runtime_spec(&raw).map(Some),
}
}
/// `ARCHIVR_DENO`, if it points to an existing file.
pub fn pinned_deno() -> Option<PathBuf> {
env::var_os(DENO_ENV)
.filter(|v| !v.is_empty())
.map(PathBuf::from)
.filter(|p| p.is_file())
}
/// Deno slot inside the state dir (`<state_dir>/deno/deno`); not existence-filtered.
pub fn state_dir_deno() -> Option<PathBuf> {
state_dir().map(|d| d.join("deno").join("deno"))
}
/// First `dir/name` that is a file, scanning `path_var` like a shell would.
pub fn find_on_path(name: &str, path_var: Option<&OsStr>) -> Option<PathBuf> {
env::split_paths(path_var?)
.filter(|dir| !dir.as_os_str().is_empty())
.map(|dir| dir.join(name))
.find(|candidate| candidate.is_file())
}
/// `deno` on the process `PATH`.
pub fn path_deno() -> Option<PathBuf> {
find_on_path("deno", env::var_os("PATH").as_deref())
}
/// Parses `deno --version` output (`deno 2.9.7 (stable, release, ...)` on the first line).
pub fn parse_deno_version_output(stdout: &str) -> Option<DenoVersion> {
let first = stdout.lines().next()?.trim();
let rest = first.strip_prefix("deno ")?;
DenoVersion::parse(rest.split_whitespace().next()?)
}
/// Runs `<binary> --version` and parses it; `None` on spawn failure, non-zero exit or junk output.
pub fn probe_deno_version(binary: &Path) -> Option<DenoVersion> {
let output = Command::new(binary).arg("--version").output().ok()?;
if !output.status.success() {
return None;
}
parse_deno_version_output(&String::from_utf8_lossy(&output.stdout))
}
/// Human version string for a (typically forced) runtime. Deno is probed and parsed; other
/// runtimes with a path report the first stdout line of `--version`; pathless non-Deno → `None`.
/// A directory path gets the runtime's executable name joined, as yt-dlp does.
pub fn probe_js_runtime_version(rt: &JsRuntime) -> Option<String> {
let binary = match &rt.path {
Some(p) if p.is_dir() => p.join(rt.kind.executable_name()),
Some(p) => p.clone(),
None if rt.kind == JsRuntimeKind::Deno => PathBuf::from("deno"),
None => return None,
};
if rt.kind == JsRuntimeKind::Deno {
return probe_deno_version(&binary).map(|v| v.to_string());
}
let output = Command::new(&binary).arg("--version").output().ok()?;
if !output.status.success() {
return None;
}
String::from_utf8_lossy(&output.stdout)
.lines()
.next()
.map(|l| l.trim().to_string())
.filter(|l| !l.is_empty())
}
/// Which candidate slot a resolved runtime came from. Several slots can point at the same
/// binary (the Nix wrappers set `ARCHIVR_DENO` and also put that Deno on `PATH`), so callers
/// that need to name the winner must use the role, not compare paths.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum JsRuntimeRole {
/// `ARCHIVR_JS_RUNTIME`.
Forced,
/// `ARCHIVR_DENO`.
Pinned,
/// `<state_dir>/deno/deno`.
StateDir,
/// `deno` on `PATH`.
Path,
}
impl JsRuntimeRole {
/// Row label used by `archivr yt-dlp status`.
pub fn label(self) -> &'static str {
match self {
Self::Forced => "force (ARCHIVR_JS_RUNTIME)",
Self::Pinned => "env (ARCHIVR_DENO)",
Self::StateDir => "state-dir",
Self::Path => "path (deno)",
}
}
/// Stable machine key (API `role`): "force" | "env" | "state-dir" | "path".
pub fn key(self) -> &'static str {
match self {
Self::Forced => "force",
Self::Pinned => "env",
Self::StateDir => "state-dir",
Self::Path => "path",
}
}
}
/// Existing Deno candidates, pinned first and state-dir last (so ties go to the state dir).
pub fn deno_candidates() -> Vec<(JsRuntimeRole, PathBuf)> {
[
(JsRuntimeRole::Pinned, pinned_deno()),
(JsRuntimeRole::StateDir, state_dir_deno()),
]
.into_iter()
.filter_map(|(role, p)| p.filter(|p| p.is_file()).map(|p| (role, p)))
.collect()
}
pub(crate) fn resolve_js_runtime_with_path(
path_var: Option<&OsStr>,
) -> Option<(JsRuntimeRole, JsRuntime)> {
if let Ok(Some(rt)) = forced_js_runtime() {
return Some((JsRuntimeRole::Forced, rt));
}
let usable = |p: &Path| probe_deno_version(p).filter(|v| *v >= MIN_DENO_VERSION);
// `max_by` keeps the last maximum, so an exact tie goes to the state dir.
let (role, best) = deno_candidates()
.into_iter()
.filter_map(|(role, p)| usable(&p).map(|v| (v, role, p)))
.max_by(|(a, ..), (b, ..)| a.cmp(b))
.map(|(_, role, p)| (role, p))
.or_else(|| {
find_on_path("deno", path_var)
.filter(|p| usable(p).is_some())
.map(|p| (JsRuntimeRole::Path, p))
})?;
Some((role, JsRuntime { kind: JsRuntimeKind::Deno, path: Some(best) }))
}
/// Resolves without caching or printing, also reporting which candidate slot won
/// (used by `archivr yt-dlp status` to mark exactly one row).
pub fn resolve_js_runtime_with_role() -> Option<(JsRuntimeRole, JsRuntime)> {
resolve_js_runtime_with_path(env::var_os("PATH").as_deref())
}
/// Cached until [`refresh_js_runtime`]; owned clone so a refresh never invalidates a caller.
pub fn resolve_js_runtime() -> Option<JsRuntime> {
if let Some(v) = RESOLVED_JS_RUNTIME.read().unwrap_or_else(|e| e.into_inner()).as_ref() {
return v.clone();
}
let mut slot = RESOLVED_JS_RUNTIME.write().unwrap_or_else(|e| e.into_inner());
slot.get_or_insert_with(resolve_js_runtime_logged).clone()
}
/// Re-resolves (outside the lock) and swaps the cache; called after a successful Deno
/// install. Commands already built keep their old runtime.
pub fn refresh_js_runtime() -> Option<JsRuntime> {
let fresh = resolve_js_runtime_logged();
*RESOLVED_JS_RUNTIME.write().unwrap_or_else(|e| e.into_inner()) = Some(fresh.clone());
fresh
}
/// yt-dlp arguments selecting `runtime`.
///
/// yt-dlp builds its runtime map keyed by name, splitting each `--js-runtimes` value on the
/// first `:` (`yt_dlp/__init__.py:784-786`), so a later entry for the same name wins:
/// `deno:<path>` replaces the default `deno` entry. Non-Deno runtimes are preceded by
/// `--no-js-runtimes` (`options.py:460-479`) so a Deno yt-dlp finds on its own can't take
/// priority over the forced choice. Each value is one argv element; no shell is involved.
pub fn js_runtime_args(runtime: Option<&JsRuntime>) -> Vec<OsString> {
let Some(rt) = runtime else {
return Vec::new();
};
let mut args = Vec::with_capacity(3);
if rt.kind != JsRuntimeKind::Deno {
args.push(OsString::from("--no-js-runtimes"));
}
args.push(OsString::from("--js-runtimes"));
args.push(rt.spec());
args
}
#[cfg(test)]
mod tests {
use super::*;
use crate::downloader::ytdlp::STATE_DIR_ENV;
use std::{fs, sync::MutexGuard};
use tempfile::TempDir;
/// Serialises env-mutating tests and clears every var the resolver reads.
fn env_guard() -> MutexGuard<'static, ()> {
let guard = crate::downloader::RESOLVER_ENV_LOCK
.lock()
.unwrap_or_else(|e| e.into_inner());
for key in [JS_RUNTIME_ENV, DENO_ENV, STATE_DIR_ENV] {
unsafe { env::remove_var(key) };
}
guard
}
/// Points the state dir at an (initially empty) dir under `tmp` and returns it.
fn set_state(tmp: &TempDir) -> PathBuf {
let state = tmp.path().join("state");
fs::create_dir_all(&state).unwrap();
unsafe { env::set_var(STATE_DIR_ENV, &state) };
state
}
/// Writes a fake `deno` script to `path` (every caller uses a fresh path), then waits until
/// it can be exec'd. A child forked by a parallel test while our write fd was open keeps a
/// copy of it until that child execs, so our exec can fail with ETXTBSY
/// (rust-lang/rust#114554). One successful exec proves no writer is left, and none can
/// appear later because our fd is already closed.
fn fake_deno(path: &Path, ver: &str) {
use std::os::unix::fs::PermissionsExt;
fs::create_dir_all(path.parent().unwrap()).unwrap();
fs::write(
path,
format!("#!/bin/sh\necho 'deno {ver} (stable, release, test)'\necho 'v8 1.0'\n"),
)
.unwrap();
fs::set_permissions(path, fs::Permissions::from_mode(0o755)).unwrap();
for _ in 0..200 {
match Command::new(path).arg("--version").output() {
Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy => {
std::thread::sleep(std::time::Duration::from_millis(5));
}
_ => return,
}
}
panic!("{} stayed busy (ETXTBSY)", path.display());
}
fn deno_at(role: JsRuntimeRole, p: PathBuf) -> Option<(JsRuntimeRole, JsRuntime)> {
Some((role, JsRuntime { kind: JsRuntimeKind::Deno, path: Some(p) }))
}
#[test]
fn spec_parsing_accepts_allowlisted_runtimes() {
let tmp = TempDir::new().unwrap();
let file = tmp.path().join("node");
fs::write(&file, "").unwrap();
assert_eq!(
parse_js_runtime_spec("deno"),
Ok(JsRuntime { kind: JsRuntimeKind::Deno, path: None })
);
assert_eq!(parse_js_runtime_spec(" NODE ").unwrap().kind, JsRuntimeKind::Node);
assert_eq!(
parse_js_runtime_spec(&format!("node:{}", file.display())),
Ok(JsRuntime { kind: JsRuntimeKind::Node, path: Some(file.clone()) })
);
assert_eq!(
parse_js_runtime_spec(&format!("deno:{}", tmp.path().display())).unwrap().path,
Some(tmp.path().to_path_buf())
);
assert_eq!(parse_js_runtime_spec("bun").unwrap().kind, JsRuntimeKind::Bun);
assert_eq!(parse_js_runtime_spec("quickjs").unwrap().kind, JsRuntimeKind::QuickJs);
}
#[test]
fn spec_parsing_rejects_bad_values() {
assert_eq!(parse_js_runtime_spec("node:"), Err("empty path after ':'".into()));
assert_eq!(parse_js_runtime_spec("node:rel/path"), Err("path must be absolute".into()));
assert_eq!(
parse_js_runtime_spec("deno:/does/not/exist"),
Err("path does not exist".into())
);
for bad in ["python", "--exec", "deno,node"] {
let err = parse_js_runtime_spec(bad).unwrap_err();
assert!(err.starts_with("unknown runtime"), "{bad}: {err}");
}
}
#[test]
fn args_follow_runtime_kind() {
assert!(js_runtime_args(None).is_empty());
let deno = JsRuntime { kind: JsRuntimeKind::Deno, path: Some("/p".into()) };
assert_eq!(js_runtime_args(Some(&deno)), ["--js-runtimes", "deno:/p"]);
let node = JsRuntime { kind: JsRuntimeKind::Node, path: Some("/p".into()) };
assert_eq!(
js_runtime_args(Some(&node)),
["--no-js-runtimes", "--js-runtimes", "node:/p"]
);
let bun = JsRuntime { kind: JsRuntimeKind::Bun, path: None };
assert_eq!(js_runtime_args(Some(&bun)), ["--no-js-runtimes", "--js-runtimes", "bun"]);
}
#[test]
fn version_parsing() {
let v297 = Some(DenoVersion { major: 2, minor: 9, patch: 7 });
assert_eq!(
parse_deno_version_output(
"deno 2.9.7 (stable, release, aarch64-apple-darwin)\nv8 14.0\ntypescript 5.9\n"
),
v297
);
assert_eq!(parse_deno_version_output("deno 2.9.7+abc123 (canary, x)\n"), v297);
assert_eq!(parse_deno_version_output("node v22"), None);
assert_eq!(parse_deno_version_output(""), None);
assert_eq!(parse_deno_version_output("deno"), None);
assert_eq!(DenoVersion::parse("v2.9.7"), v297);
assert_eq!(DenoVersion::parse("2.9"), None);
assert_eq!(DenoVersion::parse("2.9.7.1"), None);
assert!(DenoVersion::parse("2.10.0") > v297);
assert_eq!(v297.unwrap().to_string(), "2.9.7");
}
#[test]
fn newer_state_dir_deno_wins() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
let state = set_state(&tmp);
let pinned = tmp.path().join("pin/deno");
fake_deno(&pinned, "2.4.0");
fake_deno(&state.join("deno/deno"), "2.9.7");
unsafe { env::set_var(DENO_ENV, &pinned) };
assert_eq!(
resolve_js_runtime_with_path(None),
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
);
}
#[test]
fn newer_pinned_deno_wins() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
let state = set_state(&tmp);
let pinned = tmp.path().join("pin/deno");
fake_deno(&pinned, "2.10.0");
fake_deno(&state.join("deno/deno"), "2.9.7");
unsafe { env::set_var(DENO_ENV, &pinned) };
assert_eq!(
resolve_js_runtime_with_path(None),
deno_at(JsRuntimeRole::Pinned, pinned)
);
}
#[test]
fn tie_goes_to_state_dir() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
let state = set_state(&tmp);
let pinned = tmp.path().join("pin/deno");
fake_deno(&pinned, "2.9.7");
fake_deno(&state.join("deno/deno"), "2.9.7");
unsafe { env::set_var(DENO_ENV, &pinned) };
assert_eq!(
resolve_js_runtime_with_path(None),
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
);
}
#[test]
fn too_old_pinned_falls_back_to_path() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
set_state(&tmp);
let pinned = tmp.path().join("pin/deno");
fake_deno(&pinned, "2.2.9");
let bin = tmp.path().join("bin");
fake_deno(&bin.join("deno"), "2.4.0");
unsafe { env::set_var(DENO_ENV, &pinned) };
assert_eq!(
resolve_js_runtime_with_path(Some(bin.as_os_str())),
deno_at(JsRuntimeRole::Path, bin.join("deno"))
);
}
#[test]
fn valid_pinned_beats_newer_path_deno() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
set_state(&tmp);
let pinned = tmp.path().join("pin/deno");
fake_deno(&pinned, "2.4.0");
let bin = tmp.path().join("bin");
fake_deno(&bin.join("deno"), "2.9.7");
unsafe { env::set_var(DENO_ENV, &pinned) };
assert_eq!(
resolve_js_runtime_with_path(Some(bin.as_os_str())),
deno_at(JsRuntimeRole::Pinned, pinned)
);
}
#[test]
fn pinned_deno_also_on_path_wins_as_pinned_only() {
// The Nix wrappers set ARCHIVR_DENO and put the same Deno on PATH.
let _g = env_guard();
let tmp = TempDir::new().unwrap();
set_state(&tmp);
let bin = tmp.path().join("bin");
let deno = bin.join("deno");
fake_deno(&deno, "2.9.4");
unsafe { env::set_var(DENO_ENV, &deno) };
assert_eq!(find_on_path("deno", Some(bin.as_os_str())), Some(deno.clone()));
assert_eq!(
resolve_js_runtime_with_path(Some(bin.as_os_str())),
deno_at(JsRuntimeRole::Pinned, deno)
);
}
#[test]
fn forced_pinned_deno_wins_as_forced_only() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
set_state(&tmp);
let bin = tmp.path().join("bin");
let deno = bin.join("deno");
fake_deno(&deno, "2.9.4");
unsafe {
env::set_var(DENO_ENV, &deno);
env::set_var(JS_RUNTIME_ENV, format!("deno:{}", deno.display()));
}
assert_eq!(
resolve_js_runtime_with_path(Some(bin.as_os_str())),
deno_at(JsRuntimeRole::Forced, deno)
);
}
#[test]
fn too_old_path_deno_resolves_none() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
set_state(&tmp);
let bin = tmp.path().join("bin");
fake_deno(&bin.join("deno"), "2.2.9");
assert_eq!(resolve_js_runtime_with_path(Some(bin.as_os_str())), None);
}
#[test]
fn valid_forced_runtime_wins() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
let state = set_state(&tmp);
let pinned = tmp.path().join("pin/deno");
fake_deno(&pinned, "2.9.7");
fake_deno(&state.join("deno/deno"), "2.9.7");
let node = tmp.path().join("node");
fs::write(&node, "").unwrap();
unsafe {
env::set_var(DENO_ENV, &pinned);
env::set_var(JS_RUNTIME_ENV, format!("node:{}", node.display()));
}
assert_eq!(
resolve_js_runtime_with_path(None),
Some((
JsRuntimeRole::Forced,
JsRuntime { kind: JsRuntimeKind::Node, path: Some(node) }
))
);
}
#[test]
fn invalid_forced_runtime_is_ignored() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
let state = set_state(&tmp);
fake_deno(&state.join("deno/deno"), "2.9.7");
unsafe { env::set_var(JS_RUNTIME_ENV, "python") };
assert!(forced_js_runtime().is_err());
assert_eq!(
resolve_js_runtime_with_path(None),
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
);
}
#[test]
fn no_candidates_resolves_none() {
let _g = env_guard();
let tmp = TempDir::new().unwrap();
set_state(&tmp);
let empty_bin = tmp.path().join("bin");
fs::create_dir_all(&empty_bin).unwrap();
assert!(deno_candidates().is_empty());
assert_eq!(resolve_js_runtime_with_path(Some(empty_bin.as_os_str())), None);
}
#[test]
fn forced_directory_probe_joins_runtime_name() {
let tmp = TempDir::new().unwrap();
fake_deno(&tmp.path().join("deno"), "2.9.7");
let rt = JsRuntime { kind: JsRuntimeKind::Deno, path: Some(tmp.path().to_path_buf()) };
assert_eq!(probe_js_runtime_version(&rt).as_deref(), Some("2.9.7"));
}
#[test]
fn find_on_path_scans_in_order() {
let tmp = TempDir::new().unwrap();
let (a, b) = (tmp.path().join("a"), tmp.path().join("b"));
fs::create_dir_all(&a).unwrap();
fs::create_dir_all(&b).unwrap();
fs::write(b.join("tool"), "").unwrap();
let path_var = env::join_paths([&a, &b]).unwrap();
assert_eq!(find_on_path("tool", Some(&path_var)), Some(b.join("tool")));
assert_eq!(find_on_path("missing", Some(&path_var)), None);
}
#[test]
fn refresh_js_runtime_swaps_cached_choice() {
let _g = env_guard();
unsafe { env::set_var(JS_RUNTIME_ENV, "node") };
assert_eq!(refresh_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
assert_eq!(resolve_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
unsafe { env::set_var(JS_RUNTIME_ENV, "bun") };
assert_eq!(resolve_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
assert_eq!(refresh_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Bun));
unsafe { env::remove_var(JS_RUNTIME_ENV) };
refresh_js_runtime();
}
}

View file

@ -8,3 +8,32 @@ pub mod http;
pub mod singlefile;
pub mod font_extractor;
pub mod text;
pub mod js_runtime;
pub mod deno_install;
pub mod ytdlp_tools;
/// Env vars are process-global; every core test that sets resolver env vars takes this lock.
#[cfg(test)]
pub(crate) static RESOLVER_ENV_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
/// Writes an executable script to `path` (callers must use a fresh path each time), then
/// waits until it can be exec'd. A child forked by a parallel test while our write fd was
/// open keeps a copy of it until that child execs, so our own exec can fail with ETXTBSY
/// (rust-lang/rust#114554). One exec that isn't ETXTBSY proves no writer is left, and none
/// can appear later because our fd is already closed.
#[cfg(all(test, unix))]
pub(crate) fn write_script(path: &std::path::Path, body: &str) {
use std::os::unix::fs::PermissionsExt;
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
std::fs::write(path, body).unwrap();
std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o755)).unwrap();
for _ in 0..200 {
match std::process::Command::new(path).arg("--version").output() {
Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy => {
std::thread::sleep(std::time::Duration::from_millis(5));
}
_ => return,
}
}
panic!("{} stayed busy (ETXTBSY)", path.display());
}

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,559 @@
//! yt-dlp + Deno self-update and status shared by `archivr yt-dlp` and the admin API.
//! Sync; network via blocking reqwest.
use anyhow::{anyhow, bail, Context, Result};
use serde::Serialize;
use std::{
env,
path::{Path, PathBuf},
process::Command,
};
use super::deno_install;
use super::js_runtime::{
forced_js_runtime, path_deno, pinned_deno, probe_deno_version, probe_js_runtime_version,
refresh_js_runtime, resolve_js_runtime_with_role, state_dir_deno, JsRuntimeRole,
JS_RUNTIME_ENV,
};
use super::ytdlp::{
forced_yt_dlp, pinned_yt_dlp, probe_version, refresh_yt_dlp, resolve_yt_dlp, state_dir,
state_dir_yt_dlp,
};
/// GitHub release metadata endpoint for the upstream yt-dlp project.
pub const YT_DLP_LATEST_RELEASE: &str =
"https://api.github.com/repos/yt-dlp/yt-dlp/releases/latest";
/// Every python zipapp starts with this shebang; used as a sanity check that we
/// downloaded the artifact and not an HTML error page or an LFS pointer.
const ZIPAPP_SHEBANG: &[u8] = b"#!/usr/bin/env python3";
/// One candidate slot of a `status` table.
#[derive(Debug, Clone, Serialize)]
pub struct ToolCandidate {
/// Stable key: "force" | "env" | "state-dir" | "path".
pub role: &'static str,
/// Exact CLI row label.
pub label: &'static str,
/// `None` = empty slot (the CLI renders dashes).
pub path: Option<String>,
pub version: Option<String>,
pub chosen: bool,
/// Why the candidate can't be used: an invalid `ARCHIVR_JS_RUNTIME`, or a yt-dlp
/// candidate that exists but whose `--version` probe fails (last stderr line).
pub invalid: Option<String>,
}
/// The candidate the resolver picked.
#[derive(Debug, Clone, Serialize)]
pub struct ChosenTool {
/// `None` if the cached yt-dlp path matches no row.
pub role: Option<&'static str>,
/// JS only: "deno" | "node" | "bun" | "quickjs".
pub kind: Option<&'static str>,
pub path: Option<String>,
pub version: Option<String>,
}
/// Everything `archivr yt-dlp status` prints, as data.
#[derive(Debug, Clone, Serialize)]
pub struct ToolsStatus {
/// force, env, state-dir, path-fallback (CLI order).
pub yt_dlp: Vec<ToolCandidate>,
pub yt_dlp_chosen: ChosenTool,
/// force, env (ARCHIVR_DENO), state-dir, path (deno).
pub js_runtime: Vec<ToolCandidate>,
pub js_runtime_chosen: Option<ChosenTool>,
pub state_dir: Option<String>,
pub yt_dlp_target: Option<String>,
pub yt_dlp_installed: bool,
pub deno_target: Option<String>,
pub deno_installed: bool,
}
/// Per-component outcome of [`update_tools`]; `Ok` carries a one-line human outcome.
pub struct UpdateReport {
pub yt_dlp: Result<String>,
pub deno: Result<String>,
}
impl UpdateReport {
/// Names of the failed components, in `["yt-dlp", "deno"]` order.
pub fn failed_components(&self) -> Vec<&'static str> {
[("yt-dlp", self.yt_dlp.is_err()), ("deno", self.deno.is_err())]
.into_iter()
.filter_map(|(name, failed)| failed.then_some(name))
.collect()
}
}
/// Resolves `<state_dir>/yt-dlp/`, erroring out if there is no usable HOME.
fn yt_dlp_state_dir() -> Result<PathBuf> {
state_dir()
.map(|d| d.join("yt-dlp"))
.context("could not determine a state directory (is $HOME set?)")
}
/// Asks the GitHub API for the newest yt-dlp release tag.
fn latest_yt_dlp_version(client: &reqwest::blocking::Client) -> Result<String> {
let body = client
.get(YT_DLP_LATEST_RELEASE)
.send()
.context("failed to reach the GitHub releases API")?
.error_for_status()
.context("GitHub releases API returned an error")?
.text()
.context("failed to read the GitHub releases API response")?;
let json: serde_json::Value =
serde_json::from_str(&body).context("GitHub releases API returned invalid JSON")?;
json.get("tag_name")
.and_then(|t| t.as_str())
.map(str::to_string)
.context("GitHub releases API response had no tag_name")
}
/// Installs or updates the yt-dlp zipapp in the state dir. Progress goes to `log`;
/// `Ok` carries a one-line human outcome.
pub fn install_yt_dlp(
client: &reqwest::blocking::Client,
requested_version: Option<&str>,
log: &mut dyn FnMut(&str),
) -> Result<String> {
let dir = yt_dlp_state_dir()?;
let target = dir.join("yt-dlp");
let staging = dir.join("yt-dlp.new");
let version_file = dir.join(".version");
let version = match requested_version {
Some(v) => v.to_string(),
None => latest_yt_dlp_version(client)?,
};
// The sibling .version file is what lets us skip a ~3MB download on a
// no-op update; the binary itself is a zipapp with no cheap version probe
// that doesn't cost a python startup.
let installed = std::fs::read_to_string(&version_file).ok();
if target.is_file() && installed.as_deref().map(str::trim) == Some(version.as_str()) {
log(&format!("yt-dlp {version} is already installed at {}", target.display()));
return Ok(format!("yt-dlp {version} already installed at {}", target.display()));
}
log(&format!("Downloading yt-dlp {version}…"));
let url = format!("https://github.com/yt-dlp/yt-dlp/releases/download/{version}/yt-dlp");
let bytes = client
.get(&url)
.send()
.with_context(|| format!("failed to download {url}"))?
.error_for_status()
.with_context(|| format!("download failed — is {version} a real release tag?"))?
.bytes()
.context("failed to read the downloaded yt-dlp body")?;
if !bytes.starts_with(ZIPAPP_SHEBANG) {
bail!(
"downloaded artifact from {url} is not a python zipapp \
(expected it to start with `{}`) — refusing to install it",
String::from_utf8_lossy(ZIPAPP_SHEBANG)
);
}
std::fs::create_dir_all(&dir)
.with_context(|| format!("failed to create {}", dir.display()))?;
std::fs::write(&staging, &bytes)
.with_context(|| format!("failed to write {}", staging.display()))?;
#[cfg(unix)]
{
use std::os::unix::fs::PermissionsExt;
std::fs::set_permissions(&staging, std::fs::Permissions::from_mode(0o755))
.with_context(|| format!("failed to chmod +x {}", staging.display()))?;
}
// Atomic swap: a concurrently-running archivr sees either the whole old
// binary or the whole new one, never a half-written file.
std::fs::rename(&staging, &target)
.with_context(|| format!("failed to install {}", target.display()))?;
std::fs::write(&version_file, format!("{version}\n"))
.with_context(|| format!("failed to record version in {}", version_file.display()))?;
// The zipapp is python source, not a native binary — installing it on a
// host without python3 is legal (the server may run under a nix wrapper
// with its own PATH) but worth flagging loudly. Kept as `warning:` (not
// `warn:`) so the CLI's stderr is unchanged.
let has_python = Command::new("python3")
.arg("--version")
.output()
.map(|o| o.status.success())
.unwrap_or(false);
if !has_python {
eprintln!(
"warning: python3 was not found on PATH — the yt-dlp zipapp just installed \
at {} will not run until python3 is available",
target.display()
);
}
let python_note = (!has_python)
.then_some("; warning: python3 not found on PATH — the zipapp will not run until it is");
log(&format!("Installed yt-dlp {version} to {}", target.display()));
log("archivr will now prefer it whenever it is newer than the pinned binary (ARCHIVR_YT_DLP).");
Ok(format!(
"installed yt-dlp {version} to {}{}",
target.display(),
python_note.unwrap_or("")
))
}
/// Hint appended when the installed zipapp does not run; the zipapp is python source.
const PYTHON_HINT: &str = "yt-dlp needs Python ≥ 3.10 on the server's PATH as `python3`";
/// Installs yt-dlp and Deno independently: a Deno failure never blocks the yt-dlp
/// update (and vice versa). `Err` only if the HTTP client cannot be built.
///
/// With `refresh` (long-running server), the installed yt-dlp is probed with
/// `--version` — an install that does not run is reported as a failure naming the
/// cause — and each successful component refreshes its resolver cache, so the next
/// yt-dlp call uses the new binary. The one-shot CLI passes `false`: it has no cache
/// worth refreshing, and its output and probe count stay as before.
pub fn update_tools(
requested_yt_dlp_version: Option<&str>,
user_agent: &str,
refresh: bool,
log: &mut dyn FnMut(&str),
) -> Result<UpdateReport> {
let client = reqwest::blocking::Client::builder()
.user_agent(user_agent)
.build()
.context("failed to build an HTTP client")?;
let mut yt_dlp = install_yt_dlp(&client, requested_yt_dlp_version, log);
if refresh && yt_dlp.is_ok() {
if let Some(target) = state_dir_yt_dlp() {
if let Err(Some(reason)) = probe_version_detail(&target) {
yt_dlp = Err(anyhow!(unusable_install_message(&target, &reason)));
}
}
refresh_yt_dlp();
}
let deno = deno_install::install_deno(&client, log);
if refresh && deno.is_ok() {
refresh_js_runtime();
}
Ok(UpdateReport { yt_dlp, deno })
}
/// Error text for an installed yt-dlp whose `--version` probe failed.
fn unusable_install_message(target: &Path, reason: &str) -> String {
format!(
"installed {} but it does not run: {reason} — {PYTHON_HINT}",
target.display()
)
}
/// Short reason for a failed `--version` run: the last non-empty stderr line (a Python
/// traceback ends with the actual error), else the exit status.
fn probe_failure_reason(stderr: &[u8], status: &str) -> String {
String::from_utf8_lossy(stderr)
.lines()
.map(str::trim)
.rfind(|l| !l.is_empty())
.map_or_else(|| format!("--version failed ({status})"), str::to_string)
}
/// Runs `<binary> --version`. `Err(None)` = nothing to run (not found); `Err(Some(reason))`
/// = the binary exists but the probe failed.
fn probe_version_detail(binary: &Path) -> std::result::Result<String, Option<String>> {
let out = match Command::new(binary).arg("--version").output() {
Ok(out) => out,
Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Err(None),
Err(e) => return Err(Some(format!("could not run: {e}"))),
};
if !out.status.success() {
return Err(Some(probe_failure_reason(&out.stderr, &out.status.to_string())));
}
let version = String::from_utf8_lossy(&out.stdout).trim().to_string();
if version.is_empty() {
return Err(Some("--version printed nothing".into()));
}
Ok(version)
}
fn display(p: &Path) -> String {
p.display().to_string()
}
/// One yt-dlp row, probing the candidate's version; a failing probe sets `invalid`.
fn yt_row(role: &'static str, label: &'static str, path: Option<&Path>, chosen: &Path) -> ToolCandidate {
let (version, invalid) = match path.map(probe_version_detail) {
Some(Ok(v)) => (Some(v), None),
Some(Err(reason)) => (None, reason),
None => (None, None),
};
ToolCandidate {
role,
label,
path: path.map(display),
version,
chosen: path == Some(chosen),
invalid,
}
}
/// Every yt-dlp and JS runtime candidate, its version, and which one wins — the data
/// `archivr yt-dlp status` prints. Spawns `--version` probes; call off async threads.
pub fn tools_status() -> ToolsStatus {
// yt-dlp: the cached choice, i.e. what this process actually runs.
let chosen = resolve_yt_dlp();
let state_candidate = state_dir_yt_dlp().filter(|p| p.is_file());
let yt_dlp = vec![
yt_row("force", "force (ARCHIVR_YT_DLP_FORCE)", forced_yt_dlp().as_deref(), &chosen),
yt_row("env", "env (ARCHIVR_YT_DLP)", pinned_yt_dlp().as_deref(), &chosen),
yt_row("state-dir", "state-dir", state_candidate.as_deref(), &chosen),
yt_row("path", "path-fallback (yt-dlp)", Some(Path::new("yt-dlp")), &chosen),
];
let yt_dlp_chosen = match yt_dlp.iter().find(|c| c.chosen) {
Some(c) => ChosenTool {
role: Some(c.role),
kind: None,
path: Some(display(&chosen)),
version: c.version.clone(),
},
None => ChosenTool {
role: None,
kind: None,
path: Some(display(&chosen)),
version: probe_version(&chosen),
},
};
// JS runtime: uncached and silent, so status never prints the resolver warnings.
let js_chosen = resolve_js_runtime_with_role();
let chosen_role = js_chosen.as_ref().map(|(role, _)| *role);
let force_role = JsRuntimeRole::Forced;
let force_row = match forced_js_runtime() {
Ok(Some(rt)) => ToolCandidate {
role: force_role.key(),
label: force_role.label(),
path: Some(rt.spec().to_string_lossy().into_owned()),
version: probe_js_runtime_version(&rt),
chosen: chosen_role == Some(force_role),
invalid: None,
},
Ok(None) => ToolCandidate {
role: force_role.key(),
label: force_role.label(),
path: None,
version: None,
chosen: false,
invalid: None,
},
Err(reason) => ToolCandidate {
role: force_role.key(),
label: force_role.label(),
path: Some(
env::var_os(JS_RUNTIME_ENV)
.unwrap_or_default()
.to_string_lossy()
.into_owned(),
),
version: None,
chosen: false,
invalid: Some(reason),
},
};
let deno_row = |role: JsRuntimeRole, path: Option<PathBuf>| ToolCandidate {
role: role.key(),
label: role.label(),
version: path
.as_deref()
.and_then(probe_deno_version)
.map(|v| v.to_string()),
path: path.as_deref().map(display),
chosen: chosen_role == Some(role),
invalid: None,
};
let js_runtime = vec![
force_row,
deno_row(JsRuntimeRole::Pinned, pinned_deno()),
deno_row(JsRuntimeRole::StateDir, state_dir_deno().filter(|p| p.is_file())),
deno_row(JsRuntimeRole::Path, path_deno()),
];
let js_runtime_chosen = js_chosen.map(|(role, rt)| ChosenTool {
role: Some(role.key()),
kind: Some(rt.kind.as_str()),
path: rt.path.as_deref().map(display),
version: js_runtime
.iter()
.find(|c| c.chosen)
.and_then(|c| c.version.clone()),
});
let yt_dlp_target = state_dir_yt_dlp();
let deno_target = state_dir_deno();
ToolsStatus {
yt_dlp,
yt_dlp_chosen,
js_runtime,
js_runtime_chosen,
state_dir: state_dir().as_deref().map(display),
yt_dlp_installed: yt_dlp_target.as_deref().is_some_and(Path::is_file),
yt_dlp_target: yt_dlp_target.as_deref().map(display),
deno_installed: deno_target.as_deref().is_some_and(Path::is_file),
deno_target: deno_target.as_deref().map(display),
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::downloader::js_runtime::DENO_ENV;
use crate::downloader::ytdlp::{STATE_DIR_ENV, YT_DLP_ENV, YT_DLP_FORCE_ENV};
use anyhow::anyhow;
const RESOLVER_ENVS: [&str; 5] =
[YT_DLP_FORCE_ENV, YT_DLP_ENV, STATE_DIR_ENV, JS_RUNTIME_ENV, DENO_ENV];
#[cfg(unix)]
#[test]
fn tools_status_reports_forced_and_invalid_override() {
let _guard = crate::downloader::RESOLVER_ENV_LOCK
.lock()
.unwrap_or_else(|e| e.into_inner());
for key in RESOLVER_ENVS {
unsafe { env::remove_var(key) };
}
let tmp = tempfile::tempdir().unwrap();
let state = tmp.path().join("state");
let forced = tmp.path().join("forced/yt-dlp");
crate::downloader::write_script(&forced, "#!/bin/sh\necho 2020.01.01\n");
unsafe {
env::set_var(STATE_DIR_ENV, &state);
env::set_var(YT_DLP_FORCE_ENV, &forced);
env::set_var(JS_RUNTIME_ENV, "python");
}
refresh_yt_dlp();
let status = tools_status();
assert_eq!(status.yt_dlp[0].role, "force");
assert!(status.yt_dlp[0].chosen);
assert_eq!(status.yt_dlp[0].version.as_deref(), Some("2020.01.01"));
assert_eq!(status.yt_dlp_chosen.role, Some("force"));
let js_force = &status.js_runtime[0];
assert!(
js_force
.invalid
.as_deref()
.is_some_and(|r| r.contains("unknown runtime python")),
"{js_force:?}"
);
assert!(!js_force.chosen);
assert!(!status.yt_dlp_installed);
assert!(status
.yt_dlp_target
.as_deref()
.is_some_and(|t| t.ends_with("yt-dlp/yt-dlp")));
let json = serde_json::to_value(&status).unwrap();
for key in ["yt_dlp", "js_runtime", "state_dir", "deno_target"] {
assert!(json.get(key).is_some(), "missing {key}");
}
for key in RESOLVER_ENVS {
unsafe { env::remove_var(key) };
}
refresh_yt_dlp();
}
#[test]
fn probe_failure_reason_prefers_last_stderr_line() {
assert_eq!(
probe_failure_reason(
b"Traceback (most recent call last):\n File \"yt_dlp/__main__.py\", line 13\n\
ImportError: You are using an unsupported version of Python. Only Python \
versions 3.10 and above are supported by yt-dlp\n\n",
"exit status: 1"
),
"ImportError: You are using an unsupported version of Python. Only Python \
versions 3.10 and above are supported by yt-dlp"
);
assert_eq!(
probe_failure_reason(b"\n boom: too old \n", "exit status: 1"),
"boom: too old"
);
assert_eq!(
probe_failure_reason(b" \n", "exit status: 2"),
"--version failed (exit status: 2)"
);
}
#[test]
fn unusable_install_message_names_cause_and_hint() {
let msg = unusable_install_message(Path::new("/s/yt-dlp/yt-dlp"), "Only Python 3.10+");
assert!(msg.contains("/s/yt-dlp/yt-dlp"), "{msg}");
assert!(msg.contains("Only Python 3.10+"), "{msg}");
assert!(msg.contains("Python ≥ 3.10"), "{msg}");
}
#[cfg(unix)]
#[test]
fn probe_version_detail_classifies_outcomes() {
let tmp = tempfile::tempdir().unwrap();
let ok = tmp.path().join("ok");
crate::downloader::write_script(&ok, "#!/bin/sh\necho 2024.01.01\n");
assert_eq!(probe_version_detail(&ok), Ok("2024.01.01".into()));
let bad = tmp.path().join("bad");
crate::downloader::write_script(
&bad,
"#!/bin/sh\necho 'Only Python versions 3.10 and above are supported' >&2\nexit 1\n",
);
assert_eq!(
probe_version_detail(&bad),
Err(Some("Only Python versions 3.10 and above are supported".into()))
);
assert_eq!(probe_version_detail(&tmp.path().join("missing")), Err(None));
}
#[cfg(unix)]
#[test]
fn tools_status_reports_unusable_candidate_reason() {
let _guard = crate::downloader::RESOLVER_ENV_LOCK
.lock()
.unwrap_or_else(|e| e.into_inner());
for key in RESOLVER_ENVS {
unsafe { env::remove_var(key) };
}
let tmp = tempfile::tempdir().unwrap();
let pinned = tmp.path().join("pinned/yt-dlp");
crate::downloader::write_script(&pinned, "#!/bin/sh\necho 'boom: too old' >&2\nexit 1\n");
unsafe {
env::set_var(STATE_DIR_ENV, tmp.path().join("state"));
env::set_var(YT_DLP_ENV, &pinned);
}
refresh_yt_dlp();
let status = tools_status();
let env_row = &status.yt_dlp[1];
assert_eq!(env_row.role, "env");
assert_eq!(env_row.version, None);
assert_eq!(env_row.invalid.as_deref(), Some("boom: too old"));
for key in RESOLVER_ENVS {
unsafe { env::remove_var(key) };
}
refresh_yt_dlp();
}
#[test]
fn failed_components_lists_only_failures() {
let report = |y: bool, d: bool| UpdateReport {
yt_dlp: if y { Ok("ok".into()) } else { Err(anyhow!("boom")) },
deno: if d { Ok("ok".into()) } else { Err(anyhow!("boom")) },
};
assert!(report(true, true).failed_components().is_empty());
assert_eq!(report(false, true).failed_components(), ["yt-dlp"]);
assert_eq!(report(true, false).failed_components(), ["deno"]);
assert_eq!(report(false, false).failed_components(), ["yt-dlp", "deno"]);
}
}

View file

@ -0,0 +1,109 @@
//! Env-var resolution helpers shared by the summary providers and the local
//! transcription engines. External tools are configured by `ARCHIVR_*` env
//! vars only, never TOML.
use anyhow::{Result, bail};
use std::{
env,
path::{Path, PathBuf},
};
/// Reads a required env var, failing with the *exact variable name* so the
/// server can hand a caller an actionable 400 rather than "not configured".
pub(crate) fn required_env(name: &str) -> Result<String> {
match env::var(name) {
Ok(v) if !v.trim().is_empty() => Ok(v),
_ => bail!("missing required environment variable: {name}"),
}
}
pub(crate) fn env_or(name: &str, default: &str) -> String {
env::var(name)
.ok()
.filter(|v| !v.trim().is_empty())
.unwrap_or_else(|| default.to_string())
}
pub(crate) fn optional_env(name: &str) -> Option<String> {
env::var(name).ok().filter(|v| !v.trim().is_empty())
}
pub(crate) fn env_timeout(name: &str, default: u64) -> u64 {
env::var(name)
.ok()
.and_then(|v| v.trim().parse::<u64>().ok())
.filter(|v| *v > 0)
.unwrap_or(default)
}
/// Resolve a CLI executable path.
///
/// Priority: `env_name` override → first `well_known_absolute` path that
/// exists → `HOME/.local/bin/<bare>` if it exists → bare name (relies on the
/// server's PATH). The macOS defaults matter for `codex`, which the ChatGPT
/// desktop app installs at `/Applications/ChatGPT.app/Contents/Resources/codex`
/// and does not add to PATH.
pub(crate) fn resolve_cli(env_name: &str, well_known_absolute: &[&str], bare: &str) -> PathBuf {
if let Some(explicit) = optional_env(env_name) {
return PathBuf::from(explicit);
}
for candidate in well_known_absolute {
let p = Path::new(candidate);
if p.is_file() {
return p.to_path_buf();
}
}
if let Some(home) = env::var_os("HOME") {
let mut p = PathBuf::from(home);
p.push(".local/bin");
p.push(bare);
if p.is_file() {
return p;
}
}
PathBuf::from(bare)
}
#[cfg(test)]
mod tests {
use super::*;
static ENV_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
const VAR: &str = "ARCHIVR_TEST_RESOLVE_CLI";
#[test]
fn resolve_cli_prefers_env_then_absolute_then_bare() {
let _guard = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
let dir = tempfile::tempdir().unwrap();
let absolute = dir.path().join("tool");
std::fs::write(&absolute, b"").unwrap();
let absolute_str = absolute.to_str().unwrap();
let bare = "archivr-test-resolve-cli-surely-not-installed";
unsafe { env::set_var(VAR, "/explicit/tool") };
assert_eq!(
resolve_cli(VAR, &[absolute_str], bare),
PathBuf::from("/explicit/tool")
);
unsafe { env::remove_var(VAR) };
assert_eq!(resolve_cli(VAR, &["/nonexistent/x", absolute_str], bare), absolute);
assert_eq!(
resolve_cli(VAR, &["/nonexistent/x"], bare),
PathBuf::from(bare)
);
}
#[test]
fn env_timeout_ignores_zero_and_garbage() {
let _guard = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
const T: &str = "ARCHIVR_TEST_ENV_TIMEOUT";
unsafe { env::set_var(T, "0") };
assert_eq!(env_timeout(T, 7), 7);
unsafe { env::set_var(T, "abc") };
assert_eq!(env_timeout(T, 7), 7);
unsafe { env::set_var(T, " 12 ") };
assert_eq!(env_timeout(T, 7), 12);
unsafe { env::remove_var(T) };
}
}

View file

@ -5,3 +5,8 @@ pub mod downloader;
pub mod hash;
pub mod twitter;
pub mod summarizer;
pub mod subtitles;
pub mod thread_title;
pub mod transcriber;
pub(crate) mod env_config;
pub(crate) mod process;

View file

@ -0,0 +1,381 @@
//! Subprocess runner with a wall-clock timeout.
//!
//! `archivr-core` deliberately has no async runtime and the tree carries no
//! `wait_timeout` dependency, so the timeout is enforced by structure: stdout
//! and stderr are drained on their own threads (a chatty child must never
//! block on a full pipe buffer), stdin is written on a third thread (a large
//! prompt can exceed the pipe buffer), and the calling thread polls
//! `try_wait` until the child exits or the deadline passes, then kills it.
use anyhow::{Context, Result, anyhow};
use std::{
ffi::OsString,
io::{Read, Write},
path::Path,
process::{Child, Command, Stdio},
sync::mpsc,
thread,
time::{Duration, Instant},
};
/// Bytes of stderr kept for diagnostics.
const STDERR_TAIL_BYTES: usize = 4096;
/// Characters of the stderr tail quoted in a non-zero-exit error.
const EXIT_ERROR_STDERR_CHARS: usize = 400;
const POLL_INTERVAL: Duration = Duration::from_millis(50);
/// How long to wait for the pipe readers after the direct child exits before
/// assuming a grandchild holds the pipes and killing the process group.
pub(crate) const READER_GRACE: Duration = Duration::from_secs(2);
/// Puts the child in its own process group (unix) so a timeout can kill the
/// whole tree, including grandchildren that inherited the output pipes.
pub(crate) fn isolate_process_group(cmd: &mut Command) {
#[cfg(unix)]
{
use std::os::unix::process::CommandExt;
cmd.process_group(0);
}
#[cfg(not(unix))]
let _ = cmd;
}
/// SIGKILLs the process group led by `pid` (spawned via
/// [`isolate_process_group`]). Best effort; a missing group (`ESRCH`) is not
/// an error. Calls `kill(2)` directly: slim runtime images ship no `kill`
/// binary.
pub(crate) fn kill_process_group(pid: u32) {
#[cfg(unix)]
{
// kill(0, ..) hits our own group and kill(-1, ..) every process we may
// signal; a pid that doesn't fit pid_t can't be a real child either.
let Ok(pgid) = libc::pid_t::try_from(pid) else {
return;
};
if pgid <= 1 {
return;
}
// SAFETY: kill(2) takes plain integers and touches no memory of ours;
// a negative pid targets the process group `pgid`.
let _ = unsafe { libc::kill(-pgid, libc::SIGKILL) };
}
#[cfg(not(unix))]
let _ = pid;
}
/// Kills the child's whole process group and reaps the direct child.
pub(crate) fn kill_tree(child: &mut Child) {
kill_process_group(child.id());
let _ = child.kill();
let _ = child.wait();
}
/// Receives a reader result after the direct child exited: waits up to
/// `min(grace, budget)`, then kills the process group (a grandchild holding
/// the pipe) and waits one more grace period. `None` if still not done.
pub(crate) fn recv_after_exit<T>(rx: &mpsc::Receiver<T>, pid: u32, budget: Duration) -> Option<T> {
if let Ok(v) = rx.recv_timeout(READER_GRACE.min(budget)) {
return Some(v);
}
kill_process_group(pid);
rx.recv_timeout(READER_GRACE).ok()
}
#[derive(Debug)]
pub(crate) struct ProcessOutput {
pub stdout: String,
/// Last 4 KiB of stderr, lossy UTF-8.
#[allow(dead_code)]
pub stderr_tail: String,
}
/// Sentinel at the root of a timeout error, so callers can recognise a timeout
/// without string matching (see [`is_process_timeout`]).
#[derive(Debug)]
pub(crate) struct ProcessTimedOut {
pub secs: u64,
}
impl std::fmt::Display for ProcessTimedOut {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
write!(f, "timed out after {}s", self.secs)
}
}
impl std::error::Error for ProcessTimedOut {}
/// True when `error` came from a [`run_with_timeout`] deadline, however many
/// context layers have been added on top since.
pub(crate) fn is_process_timeout(error: &anyhow::Error) -> bool {
error
.chain()
.find_map(|c| c.downcast_ref::<ProcessTimedOut>())
.or_else(|| error.downcast_ref::<ProcessTimedOut>())
.is_some()
}
/// Spawns `executable args…`, optionally writes `stdin`, drains stdout and
/// stderr on their own threads, and kills the child if it is still running at
/// `timeout`.
///
/// - Non-zero exit → `Err("{exe} exited with {status}: {last 400 chars of stderr}")`.
/// - Timeout → an error whose root is [`ProcessTimedOut`] with the message
/// `"{exe} timed out after {secs}s"`.
pub(crate) fn run_with_timeout(
executable: &Path,
args: &[OsString],
stdin: Option<&str>,
timeout: Duration,
) -> Result<ProcessOutput> {
let exe = executable.display().to_string();
let started = Instant::now();
let timeout_error = || {
let secs = timeout.as_secs().max(1);
anyhow::Error::new(ProcessTimedOut { secs }).context(format!("{exe} timed out after {secs}s"))
};
let mut command = Command::new(executable);
command
.args(args)
.stdin(if stdin.is_some() {
Stdio::piped()
} else {
Stdio::null()
})
.stdout(Stdio::piped())
.stderr(Stdio::piped());
isolate_process_group(&mut command);
let mut child = command
.spawn()
.with_context(|| format!("failed to spawn {exe}"))?;
let pid = child.id();
if let Some(input) = stdin {
let mut pipe = child
.stdin
.take()
.ok_or_else(|| anyhow!("failed to open stdin for {exe}"))?;
let owned = input.to_string();
thread::spawn(move || {
let _ = pipe.write_all(owned.as_bytes());
// Dropping the pipe closes it, which tells the child input is complete.
});
}
let mut stdout = child
.stdout
.take()
.ok_or_else(|| anyhow!("failed to open stdout for {exe}"))?;
let (out_tx, out_rx) = mpsc::channel();
thread::spawn(move || {
let mut buf = String::new();
let res = stdout.read_to_string(&mut buf).map(|_| buf);
let _ = out_tx.send(res);
});
let mut stderr = child
.stderr
.take()
.ok_or_else(|| anyhow!("failed to open stderr for {exe}"))?;
let (err_tx, err_rx) = mpsc::channel();
thread::spawn(move || {
let mut tail: Vec<u8> = Vec::new();
let mut chunk = [0u8; 8192];
loop {
match stderr.read(&mut chunk) {
Ok(0) | Err(_) => break,
Ok(n) => {
tail.extend_from_slice(&chunk[..n]);
if tail.len() > STDERR_TAIL_BYTES {
let excess = tail.len() - STDERR_TAIL_BYTES;
tail.drain(..excess);
}
}
}
}
let _ = err_tx.send(String::from_utf8_lossy(&tail).into_owned());
});
let status = loop {
match child.try_wait() {
Ok(Some(status)) => break status,
Ok(None) => {}
Err(e) => {
kill_tree(&mut child);
return Err(anyhow::Error::new(e).context(format!("failed to wait for {exe}")));
}
}
if started.elapsed() >= timeout {
// Kill the whole group so grandchildren release the pipes too.
kill_tree(&mut child);
return Err(timeout_error());
}
thread::sleep(POLL_INTERVAL);
};
// The child has exited, but a grandchild that inherited the pipes can keep
// them open; give the readers a short grace, then kill the group.
let remaining = || timeout.saturating_sub(started.elapsed());
let collected = match recv_after_exit(&out_rx, pid, remaining()) {
Some(res) => res.with_context(|| format!("failed to read stdout of {exe}"))?,
None => return Err(timeout_error()),
};
let Some(stderr_tail) = recv_after_exit(&err_rx, pid, remaining()) else {
return Err(timeout_error());
};
if !status.success() {
anyhow::bail!(
"{exe} exited with {status}: {}",
last_chars(stderr_tail.trim(), EXIT_ERROR_STDERR_CHARS)
);
}
Ok(ProcessOutput {
stdout: collected,
stderr_tail,
})
}
fn last_chars(s: &str, max: usize) -> String {
let count = s.chars().count();
if count <= max {
return s.to_string();
}
let tail: String = s.chars().skip(count - max).collect();
format!("…{tail}")
}
/// Shared by process-group tests here and in `downloader::ytdlp`.
#[cfg(all(test, unix))]
pub(crate) mod test_support {
use std::{path::Path, process::Command, time::{Duration, Instant}};
/// Shell snippet: start a background `sleep 30` and record its pid in `pid_file`.
pub(crate) fn spawn_grandchild_snippet(pid_file: &Path) -> String {
format!("sleep 30 & echo $! > '{}'; ", pid_file.display())
}
/// Reads the pid written by [`spawn_grandchild_snippet`] and asserts the
/// process disappears within a few seconds (allowing init to reap it).
pub(crate) fn assert_grandchild_gone(pid_file: &Path) {
let pid = std::fs::read_to_string(pid_file).unwrap().trim().to_string();
assert!(!pid.is_empty(), "grandchild pid not recorded");
let deadline = Instant::now() + Duration::from_secs(5);
loop {
let alive = Command::new("kill")
.args(["-0", &pid])
.stderr(std::process::Stdio::null())
.status()
.map(|s| s.success())
.unwrap_or(false);
if !alive {
return;
}
if Instant::now() >= deadline {
let _ = Command::new("kill").args(["-KILL", &pid]).status();
panic!("grandchild {pid} survived the timeout kill");
}
std::thread::sleep(Duration::from_millis(50));
}
}
}
#[cfg(test)]
mod tests {
use super::*;
fn os(args: &[&str]) -> Vec<OsString> {
args.iter().map(OsString::from).collect()
}
#[test]
fn run_with_timeout_drains_large_stderr_without_deadlock() {
let out = run_with_timeout(
Path::new("sh"),
&os(&["-c", "head -c 1000000 /dev/zero | tr '\\0' x >&2; echo ok"]),
None,
Duration::from_secs(30),
)
.unwrap();
assert_eq!(out.stdout, "ok\n");
assert_eq!(out.stderr_tail.len(), STDERR_TAIL_BYTES);
}
#[test]
fn run_with_timeout_kills_overrunning_child_and_marks_timeout() {
let started = Instant::now();
let err = run_with_timeout(
Path::new("sleep"),
&os(&["30"]),
None,
Duration::from_secs(1),
)
.unwrap_err();
assert!(is_process_timeout(&err), "{err:#}");
assert!(format!("{err:#}").contains("timed out after 1s"), "{err:#}");
assert!(started.elapsed() < Duration::from_secs(5));
// Still recognisable under further context layers.
let wrapped = err.context("outer").context("outermost");
assert!(is_process_timeout(&wrapped));
}
#[test]
fn run_with_timeout_reports_nonzero_exit_with_stderr_tail() {
let err = run_with_timeout(
Path::new("sh"),
&os(&["-c", "echo first-line >&2; echo boom-at-the-end >&2; exit 3"]),
None,
Duration::from_secs(30),
)
.unwrap_err();
let msg = format!("{err:#}");
assert!(msg.contains("exited with"), "{msg}");
assert!(msg.contains("boom-at-the-end"), "{msg}");
assert!(!is_process_timeout(&err));
}
#[test]
fn run_with_timeout_round_trips_stdin() {
let out = run_with_timeout(
Path::new("cat"),
&[],
Some("prompt text"),
Duration::from_secs(30),
)
.unwrap();
assert_eq!(out.stdout, "prompt text");
}
#[test]
fn last_chars_keeps_the_tail() {
assert_eq!(last_chars("abc", 5), "abc");
assert_eq!(last_chars("abcdef", 3), "…def");
}
#[cfg(unix)]
#[test]
fn run_with_timeout_kills_grandchildren_on_timeout() {
let dir = tempfile::tempdir().unwrap();
let pid_file = dir.path().join("grandchild.pid");
let script = format!("{}sleep 30", test_support::spawn_grandchild_snippet(&pid_file));
let started = Instant::now();
let err = run_with_timeout(Path::new("sh"), &os(&["-c", &script]), None, Duration::from_secs(1))
.unwrap_err();
assert!(is_process_timeout(&err), "{err:#}");
assert!(started.elapsed() < Duration::from_secs(5));
test_support::assert_grandchild_gone(&pid_file);
}
#[cfg(unix)]
#[test]
fn run_with_timeout_does_not_wait_out_budget_for_pipe_holding_grandchild() {
let dir = tempfile::tempdir().unwrap();
let pid_file = dir.path().join("grandchild.pid");
let script = format!("{}echo done", test_support::spawn_grandchild_snippet(&pid_file));
let started = Instant::now();
let out = run_with_timeout(Path::new("sh"), &os(&["-c", &script]), None, Duration::from_secs(60))
.unwrap();
assert_eq!(out.stdout, "done\n");
assert!(started.elapsed() < Duration::from_secs(10), "{:?}", started.elapsed());
test_support::assert_grandchild_gone(&pid_file);
}
}

View file

@ -0,0 +1,844 @@
//! Subtitle artifacts: archiving staged yt-dlp subtitle files, registering
//! them as `subtitle` artifacts, fetching them on demand for existing entries,
//! ranking tracks, and reducing VTT/SRT to a plain transcript for summaries.
use anyhow::{anyhow, Context, Result};
use regex::Regex;
use rusqlite::{Connection, Transaction, TransactionBehavior};
use std::{
fs,
path::{Path, PathBuf},
sync::LazyLock,
};
use uuid::Uuid;
use crate::archive::ArchivePaths;
use crate::capture;
use crate::database::{self, BlobRecord, NewArtifact};
use crate::downloader::store;
use crate::downloader::ytdlp::{self, language_base, StagedSubtitle, SubtitleKind};
pub const SUBTITLE_ARTIFACT_ROLE: &str = "subtitle";
pub const SUBTITLE_ORIGIN_CAPTURE: &str = "capture";
pub const SUBTITLE_ORIGIN_SUMMARY_FETCH: &str = "summary_fetch";
/// Origin of a track produced by local transcription (`kind: "transcribed"`).
pub const SUBTITLE_ORIGIN_TRANSCRIPTION: &str = "transcription";
/// Result of [`fetch_subtitles_for_entry`].
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct SubtitleFetchOutcome {
/// Artifact rows inserted by this call.
pub added: usize,
/// The video's original language, from a successful metadata probe or
/// else from existing subtitle artifacts' metadata.
pub original_language: Option<String>,
}
/// Subtitle file formats archivr keeps.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum SubtitleFormat {
Vtt,
Srt,
}
impl SubtitleFormat {
/// Detects the format from a file extension (with or without the dot),
/// falling back to the MIME type. Case-insensitive.
pub fn detect(extension: &str, mime: &str) -> Option<Self> {
let ext = extension.trim_start_matches('.').to_ascii_lowercase();
match ext.as_str() {
"vtt" => return Some(SubtitleFormat::Vtt),
"srt" => return Some(SubtitleFormat::Srt),
_ => {}
}
let mime = mime.split(';').next().unwrap_or("").trim().to_ascii_lowercase();
match mime.as_str() {
"text/vtt" => Some(SubtitleFormat::Vtt),
"application/x-subrip" => Some(SubtitleFormat::Srt),
_ => None,
}
}
pub fn mime(self) -> &'static str {
match self {
SubtitleFormat::Vtt => "text/vtt",
SubtitleFormat::Srt => "application/x-subrip",
}
}
pub fn extension(self) -> &'static str {
match self {
SubtitleFormat::Vtt => "vtt",
SubtitleFormat::Srt => "srt",
}
}
}
/// A subtitle file already moved into the content-addressed `raw/` store.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct ArchivedSubtitle {
/// Store-relative path, e.g. `raw/a/b/<hash>.vtt`.
pub raw_relpath: PathBuf,
pub language: String,
pub kind: SubtitleKind,
pub format: SubtitleFormat,
pub original_language: Option<String>,
}
/// Track description parsed from a `subtitle` artifact's `metadata_json`.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct SubtitleTrackMeta {
pub language: String,
pub kind: SubtitleKind,
pub original_language: Option<String>,
}
/// Moves staged subtitle files into `raw/`. Files that fail to archive or have
/// an unsupported format are logged and skipped — subtitles never fail a capture.
pub fn archive_staged_subtitles(
store_path: &Path,
staged: Vec<StagedSubtitle>,
) -> Vec<ArchivedSubtitle> {
let mut archived = Vec::with_capacity(staged.len());
for sub in staged {
let Some(format) = SubtitleFormat::detect(&sub.format, "") else {
eprintln!(
"warn: archive subtitle {}: unsupported format {}",
sub.path.display(),
sub.format
);
continue;
};
match store::archive_staged_file(&sub.path, store_path) {
Ok(raw_relpath) => archived.push(ArchivedSubtitle {
raw_relpath,
language: sub.language,
kind: sub.kind,
format,
original_language: sub.original_language,
}),
Err(e) => eprintln!("warn: archive subtitle {}: {e:#}", sub.path.display()),
}
}
archived
}
/// Registers archived subtitles as `subtitle` artifacts of `entry_id`.
///
/// Runs in one `BEGIN IMMEDIATE` transaction so concurrent registrations of
/// the same content serialize; an `(entry, subtitle, blob)` that already
/// exists is skipped. Returns the number of artifact rows inserted.
pub fn register_subtitle_artifacts(
conn: &Connection,
store_path: &Path,
entry_id: i64,
subtitles: &[ArchivedSubtitle],
origin: &str,
) -> Result<usize> {
let rows: Vec<(&ArchivedSubtitle, serde_json::Value)> = subtitles
.iter()
.map(|sub| {
let metadata = serde_json::json!({
"language": sub.language,
"kind": sub.kind.as_str(),
"format": sub.format.extension(),
"original_language": sub.original_language,
"origin": origin,
});
(sub, metadata)
})
.collect();
insert_subtitle_rows(conn, store_path, entry_id, &rows)
}
/// Registers a locally transcribed track (origin `transcription`), recording
/// the engine kind and model. A model given as a filesystem path is stored as
/// its file name only, so no host path is persisted. Same transaction and
/// dedup rules as [`register_subtitle_artifacts`].
pub fn register_transcript_artifact(
conn: &Connection,
store_path: &Path,
entry_id: i64,
sub: &ArchivedSubtitle,
engine: &str,
model: &str,
) -> Result<usize> {
let metadata = serde_json::json!({
"language": sub.language,
"kind": sub.kind.as_str(),
"format": sub.format.extension(),
"original_language": sub.original_language,
"origin": SUBTITLE_ORIGIN_TRANSCRIPTION,
"engine": engine,
"model": sanitize_model_name(model),
});
insert_subtitle_rows(conn, store_path, entry_id, &[(sub, metadata)])
}
/// Reduces a model that is a filesystem path (contains `\`, is absolute, or
/// exists) to its file name. Hugging Face ids such as
/// `nvidia/parakeet-tdt-0.6b-v3` are kept as-is.
pub(crate) fn sanitize_model_name(model: &str) -> String {
let path = Path::new(model);
if model.contains('\\') || path.is_absolute() || path.exists() {
let name = model.rsplit(['/', '\\']).next().unwrap_or(model);
if !name.is_empty() {
return name.to_string();
}
}
model.to_string()
}
/// Inserts one `subtitle` artifact per row with the given metadata, in one
/// `BEGIN IMMEDIATE` transaction; rows whose blob is already a subtitle of
/// the entry, or whose file can't be stat'ed, are skipped. Refreshes the
/// entry's cached bytes and returns the number of rows inserted.
fn insert_subtitle_rows(
conn: &Connection,
store_path: &Path,
entry_id: i64,
rows: &[(&ArchivedSubtitle, serde_json::Value)],
) -> Result<usize> {
if rows.is_empty() {
return Ok(0);
}
// No transaction is open on `conn` at any call site, so new_unchecked is safe.
let tx = Transaction::new_unchecked(conn, TransactionBehavior::Immediate)?;
let mut inserted = 0;
for (sub, metadata) in rows {
let relpath = sub.raw_relpath.to_string_lossy().replace('\\', "/");
let sha256 = sub
.raw_relpath
.file_stem()
.and_then(|s| s.to_str())
.with_context(|| format!("subtitle path has no hash stem: {relpath}"))?
.to_string();
let byte_size = match fs::metadata(store_path.join(&sub.raw_relpath)) {
Ok(meta) => meta.len() as i64,
Err(e) => {
eprintln!("warn: skipping archived subtitle {relpath}: {e:#}");
continue;
}
};
let blob_id = database::upsert_blob(
&tx,
&BlobRecord {
sha256,
byte_size,
mime_type: Some(sub.format.mime().to_string()),
extension: Some(sub.format.extension().to_string()),
raw_relpath: relpath.clone(),
},
)?;
if database::entry_has_artifact_blob(&tx, entry_id, SUBTITLE_ARTIFACT_ROLE, blob_id)? {
continue;
}
database::add_entry_artifact(
&tx,
&NewArtifact {
entry_id,
artifact_role: SUBTITLE_ARTIFACT_ROLE.to_string(),
storage_area: "raw".to_string(),
relpath,
blob_id: Some(blob_id),
logical_path: None,
metadata_json: Some(metadata.to_string()),
},
)?;
inserted += 1;
}
tx.commit()?;
database::refresh_entry_cached_bytes(conn, entry_id)?;
Ok(inserted)
}
/// Number of the entry's `subtitle` artifacts that reduce to a non-empty
/// transcript. Unreadable or unsupported files do not count.
pub(crate) fn usable_subtitle_count(
conn: &Connection,
store_path: &Path,
entry_id: i64,
) -> Result<usize> {
let artifacts = database::list_entry_artifacts_by_role(conn, entry_id, SUBTITLE_ARTIFACT_ROLE)?;
Ok(artifacts
.iter()
.filter(|a| {
let ext = Path::new(&a.relpath)
.extension()
.and_then(|e| e.to_str())
.unwrap_or("");
SubtitleFormat::detect(ext, a.mime_type.as_deref().unwrap_or("")).is_some()
&& fs::read_to_string(store_path.join(&a.relpath))
.map(|raw| !subtitle_to_transcript(&raw).is_empty())
.unwrap_or(false)
})
.count())
}
/// `original_language` of the entry's first `subtitle` artifact (id order)
/// whose metadata records one.
fn existing_original_language(conn: &Connection, entry_id: i64) -> Result<Option<String>> {
let artifacts = database::list_entry_artifacts_by_role(conn, entry_id, SUBTITLE_ARTIFACT_ROLE)?;
Ok(artifacts
.iter()
.find_map(|a| parse_subtitle_metadata(a.metadata_json.as_deref()).original_language))
}
/// Downloads subtitles for an existing YouTube video entry from its original
/// URL and registers them. Returns the number of rows added by this call plus
/// the video's original language, taken from the metadata probe when it
/// succeeds and otherwise from existing subtitle artifacts.
///
/// Returns `added: 0` (no yt-dlp call) for non-YouTube-video entries or a
/// missing / non-http(s) canonical URL, and without fetching when the entry
/// already has a usable subtitle (concurrency re-check). An unreachable video
/// or any yt-dlp failure is logged and counts as zero subtitles; only DB/IO
/// errors propagate.
pub fn fetch_subtitles_for_entry(
paths: &ArchivePaths,
entry_uid: &str,
cookie_rules: &[database::CookieRule],
) -> Result<SubtitleFetchOutcome> {
let conn = database::open_or_initialize(&paths.archive_path)?;
let info = database::entry_source_info(&conn, entry_uid)?
.ok_or_else(|| anyhow!("entry not found: {entry_uid}"))?;
let existing_language = existing_original_language(&conn, info.entry_id)?;
let nothing_added = || SubtitleFetchOutcome {
added: 0,
original_language: existing_language.clone(),
};
if info.source_kind != "youtube" || info.entity_kind != "video" {
return Ok(nothing_added());
}
let Some(url) = info
.canonical_url
.filter(|u| u.starts_with("https://") || u.starts_with("http://"))
else {
return Ok(nothing_added());
};
let store_path = &paths.store_path;
if usable_subtitle_count(&conn, store_path, info.entry_id)? > 0 {
return Ok(nothing_added());
}
let cookies = capture::resolve_cookies_for_url(cookie_rules, &url);
let timeout = crate::summarizer::summary_cli_timeout();
let Some(metadata) = ytdlp::fetch_metadata_with_timeout(&url, &cookies, Some(timeout)) else {
eprintln!("warn: subtitle fetch for {entry_uid}: video unreachable ({url})");
return Ok(nothing_added());
};
// Before planning: a video without captions is exactly the case that
// needs its language for a transcription fallback.
let original_language = serde_json::from_str::<serde_json::Value>(&metadata)
.ok()
.and_then(|v| ytdlp::original_language_from_metadata(&v))
.or_else(|| existing_language.clone());
let Some(request) = ytdlp::plan_subtitle_request(Some(&metadata)) else {
eprintln!("info: subtitle fetch for {entry_uid}: no subtitle tracks available ({url})");
return Ok(SubtitleFetchOutcome {
added: 0,
original_language,
});
};
let stage_key = format!("subs-{}", Uuid::new_v4().simple());
let stage_dir = store_path.join("temp").join(&stage_key);
let staged = match ytdlp::download_subtitles(&url, store_path, &stage_key, &request, &cookies, timeout) {
Ok(staged) => staged,
Err(e) => {
eprintln!("warn: subtitle fetch for {entry_uid} failed: {e:#}");
let _ = fs::remove_dir_all(&stage_dir);
return Ok(SubtitleFetchOutcome {
added: 0,
original_language,
});
}
};
let archived = archive_staged_subtitles(store_path, staged);
let _ = fs::remove_dir_all(&stage_dir);
let added = register_subtitle_artifacts(
&conn,
store_path,
info.entry_id,
&archived,
SUBTITLE_ORIGIN_SUMMARY_FETCH,
)?;
eprintln!("info: subtitle fetch for {entry_uid}: registered {added} subtitle artifact(s)");
Ok(SubtitleFetchOutcome {
added,
original_language,
})
}
/// Any `<...>` markup: `<c>`, `<c.colorE5E5E5>`, `<00:00:01.000>`, `<v Speaker>`, `<i>`.
static TAG_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"<[^>]*>").expect("valid tag regex"));
/// SRT ASS override blocks such as `{\an8}`.
static ASS_OVERRIDE_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"\{\\[^}]*\}").expect("valid ASS override regex"));
/// Strips markup from one cue text line and normalizes whitespace.
fn clean_cue_line(line: &str) -> String {
let without_tags = TAG_RE.replace_all(line, "");
let without_ass = ASS_OVERRIDE_RE.replace_all(&without_tags, "");
// `&amp;` last so `&amp;lt;` decodes to `&lt;`, not `<`.
let decoded = without_ass
.replace("&lt;", "<")
.replace("&gt;", ">")
.replace("&quot;", "\"")
.replace("&#39;", "'")
.replace("&nbsp;", " ")
.replace("&amp;", "&");
decoded.split_whitespace().collect::<Vec<_>>().join(" ")
}
/// Appends `line` unless it repeats one of the last two lines; a line that
/// extends the previous one (rolling auto-captions) replaces it.
fn push_deduped(out: &mut Vec<String>, line: String) {
let recent = &out[out.len().saturating_sub(2)..];
if recent.iter().any(|l| *l == line) {
return;
}
if let Some(last) = out.last_mut() {
if line.len() > last.len() && line.starts_with(last.as_str()) {
*last = line;
return;
}
}
out.push(line);
}
/// Text lines of one cue block (everything after its timing line).
fn reduce_block(block: &[&str], out: &mut Vec<String>) {
let Some(timing) = block.iter().position(|l| l.contains("-->")) else {
return; // WEBVTT header, NOTE, STYLE, REGION, bare index
};
for line in &block[timing + 1..] {
let cleaned = clean_cue_line(line);
if !cleaned.is_empty() {
push_deduped(out, cleaned);
}
}
}
/// Reduces a VTT or SRT document to plain transcript text, one line per
/// caption line, with markup, timings and rolling-caption repeats removed.
///
/// Blocks split only on truly empty lines: YouTube auto-caption cues contain
/// lines holding a single space, which belong to the cue.
pub fn subtitle_to_transcript(raw: &str) -> String {
let text = raw
.strip_prefix('\u{feff}')
.unwrap_or(raw)
.replace("\r\n", "\n")
.replace('\r', "\n");
let mut out: Vec<String> = Vec::new();
let mut block: Vec<&str> = Vec::new();
for line in text.split('\n') {
if line.is_empty() {
reduce_block(&block, &mut out);
block.clear();
} else {
block.push(line);
}
}
reduce_block(&block, &mut out);
out.join("\n")
}
/// Parses a `subtitle` artifact's `metadata_json`. Missing or invalid metadata
/// yields language `""` and `Unknown` kind.
pub fn parse_subtitle_metadata(metadata_json: Option<&str>) -> SubtitleTrackMeta {
let value: serde_json::Value = metadata_json
.and_then(|json| serde_json::from_str(json).ok())
.unwrap_or(serde_json::Value::Null);
let text = |key: &str| {
value
.get(key)
.and_then(|v| v.as_str())
.map(str::trim)
.filter(|s| !s.is_empty())
.map(str::to_string)
};
SubtitleTrackMeta {
language: text("language").unwrap_or_default(),
kind: text("kind")
.map(|k| SubtitleKind::parse(&k))
.unwrap_or(SubtitleKind::Unknown),
original_language: text("original_language"),
}
}
/// Preference rank of a subtitle track for summaries; lower is better.
///
/// 0 manual English, 1 manual original-language, 2 other manual,
/// 3 transcribed (any language), 4 auto/unknown original-language,
/// 5 auto/unknown English, 6 anything else.
pub fn subtitle_track_rank(meta: &SubtitleTrackMeta) -> u8 {
let base = language_base(&meta.language);
let is_en = base == "en";
let is_orig = meta.language.to_ascii_lowercase().ends_with("-orig")
|| (!base.is_empty()
&& meta
.original_language
.as_deref()
.is_some_and(|orig| language_base(orig) == base));
match meta.kind {
SubtitleKind::Manual if is_en => 0,
SubtitleKind::Manual if is_orig => 1,
SubtitleKind::Manual => 2,
SubtitleKind::Transcribed => 3,
_ if is_orig => 4,
_ if is_en => 5,
_ => 6,
}
}
#[cfg(test)]
mod tests {
use super::*;
fn meta(language: &str, kind: SubtitleKind, original: Option<&str>) -> SubtitleTrackMeta {
SubtitleTrackMeta {
language: language.to_string(),
kind,
original_language: original.map(str::to_string),
}
}
#[test]
fn vtt_reduction_strips_header_timestamps_settings_and_tags() {
let vtt = "WEBVTT\nKind: captions\nLanguage: en\n\nNOTE a comment\nspanning lines\n\nSTYLE\n::cue { color: red }\n\ncue-1\n00:00:01.000 --> 00:00:03.000 align:start position:0%\n<v Speaker>Hello <i>there</i></v>\n\n00:00:03.000 --> 00:00:05.000\n<c.colorE5E5E5>General</c> <00:00:03.500><c>Kenobi</c>\n";
assert_eq!(subtitle_to_transcript(vtt), "Hello there\nGeneral Kenobi");
}
#[test]
fn vtt_reduction_collapses_rolling_auto_captions() {
// Real-shaped YouTube auto-caption VTT: each cue repeats the previous
// line, 10 ms "freeze" cues duplicate it, and lines holding a single
// space sit inside cues (they must not split blocks).
let vtt = "WEBVTT\nKind: captions\nLanguage: en\n\n\
00:00:00.000 --> 00:00:02.030 align:start position:0%\n \nhello<00:00:00.320><c> world</c><00:00:00.640><c> this</c>\n\n\
00:00:02.030 --> 00:00:02.040 align:start position:0%\nhello world this\n \n\n\
00:00:02.040 --> 00:00:04.110 align:start position:0%\nhello world this\nis<00:00:02.360><c> a</c><00:00:02.600><c> test</c>\n\n\
00:00:04.110 --> 00:00:04.120 align:start position:0%\nis a test\n \n\n\
00:00:04.120 --> 00:00:06.000 align:start position:0%\nis a test\nof<00:00:04.500><c> captions</c>\n\n\
00:00:06.000 --> 00:00:06.010 align:start position:0%\nof captions\n \n";
assert_eq!(
subtitle_to_transcript(vtt),
"hello world this\nis a test\nof captions"
);
}
#[test]
fn vtt_reduction_extends_growing_lines() {
let vtt = "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nhello\n\n00:00:01.000 --> 00:00:02.000\nhello world\n";
assert_eq!(subtitle_to_transcript(vtt), "hello world");
}
#[test]
fn srt_reduction_strips_indices_italics_and_ass_overrides() {
let srt = "1\n00:00:01,000 --> 00:00:02,000\n<i>Hello</i> there\n\n2\n00:00:02,500 --> 00:00:04,000\n{\\an8}Second line\n<b>continues</b> here\n\n3\n00:00:04,000 --> 00:00:05,000\n42\n";
assert_eq!(
subtitle_to_transcript(srt),
"Hello there\nSecond line\ncontinues here\n42"
);
}
#[test]
fn reduction_handles_bom_crlf_and_entities() {
let vtt = "\u{feff}WEBVTT\r\n\r\n00:00:01.000 --> 00:00:02.000\r\nTom &amp; Jerry &lt;3&nbsp;&quot;cheese&quot; it&#39;s &amp;lt;\r\n\r\n00:00:02.000 --> 00:00:03.000\rold mac line\r";
assert_eq!(
subtitle_to_transcript(vtt),
"Tom & Jerry <3 \"cheese\" it's &lt;\nold mac line"
);
assert_eq!(subtitle_to_transcript(""), "");
assert_eq!(subtitle_to_transcript("WEBVTT\n\n"), "");
}
#[test]
fn track_rank_prefers_manual_english_then_manual_original_then_auto_original() {
let de = Some("de");
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Manual, de)), 0);
assert_eq!(subtitle_track_rank(&meta("en-GB", SubtitleKind::Manual, de)), 0);
assert_eq!(subtitle_track_rank(&meta("de", SubtitleKind::Manual, de)), 1);
assert_eq!(subtitle_track_rank(&meta("fr", SubtitleKind::Manual, de)), 2);
assert_eq!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Auto, de)), 4);
assert_eq!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Unknown, None)), 4);
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Auto, de)), 5);
assert_eq!(subtitle_track_rank(&meta("fr", SubtitleKind::Auto, de)), 6);
assert_eq!(subtitle_track_rank(&meta("", SubtitleKind::Unknown, None)), 6);
}
#[test]
fn track_rank_places_transcribed_below_manual_above_auto() {
let de = Some("de");
let transcribed = subtitle_track_rank(&meta("fr", SubtitleKind::Transcribed, de));
assert_eq!(transcribed, 3);
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Transcribed, None)), 3);
assert!(subtitle_track_rank(&meta("fr", SubtitleKind::Manual, de)) < transcribed);
assert!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Auto, de)) > transcribed);
assert!(subtitle_track_rank(&meta("en", SubtitleKind::Auto, de)) > transcribed);
}
#[test]
fn parse_subtitle_metadata_defaults_and_round_trip() {
assert_eq!(
parse_subtitle_metadata(None),
meta("", SubtitleKind::Unknown, None)
);
assert_eq!(
parse_subtitle_metadata(Some("not json")),
meta("", SubtitleKind::Unknown, None)
);
assert_eq!(
parse_subtitle_metadata(Some(
r#"{"language":"de-orig","kind":"auto","format":"vtt","original_language":"de","origin":"capture"}"#
)),
meta("de-orig", SubtitleKind::Auto, Some("de"))
);
}
#[test]
fn subtitle_format_detects_by_extension_or_mime() {
assert_eq!(SubtitleFormat::detect("vtt", ""), Some(SubtitleFormat::Vtt));
assert_eq!(SubtitleFormat::detect(".SRT", ""), Some(SubtitleFormat::Srt));
assert_eq!(
SubtitleFormat::detect("", "text/vtt; charset=utf-8"),
Some(SubtitleFormat::Vtt)
);
assert_eq!(
SubtitleFormat::detect("txt", "application/x-subrip"),
Some(SubtitleFormat::Srt)
);
assert_eq!(SubtitleFormat::detect("ttml", "application/ttml+xml"), None);
}
fn archive_fixture(
source_kind: &str,
entity_kind: &str,
canonical_url: Option<&str>,
) -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
let temp = tempfile::tempdir().unwrap();
let paths = crate::archive::initialize_archive(
temp.path(),
&temp.path().join("store"),
"Test archive",
false,
)
.unwrap();
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
let user_id = database::ensure_default_user(&conn).unwrap();
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
let source_id = database::upsert_source_identity(
&conn,
source_kind,
entity_kind,
Some("fixture-1"),
canonical_url,
canonical_url.unwrap_or("fixture:1"),
)
.unwrap();
let entry = database::create_archived_entry(
&conn,
&database::NewEntry {
source_identity_id: source_id,
archive_run_id: run.id,
parent_entry_id: None,
root_entry_id: None,
created_by_user_id: user_id,
owned_by_user_id: user_id,
source_kind: source_kind.to_string(),
entity_kind: entity_kind.to_string(),
title: None,
visibility: "private".to_string(),
representation_kind: entity_kind.to_string(),
source_metadata_json: "{}".to_string(),
display_metadata_json: None,
},
)
.unwrap();
(temp, paths, entry)
}
fn stage_vtt(store_path: &Path, name: &str, body: &str) -> StagedSubtitle {
let dir = store_path.join("temp").join("stage");
fs::create_dir_all(&dir).unwrap();
let path = dir.join(name);
fs::write(&path, body).unwrap();
StagedSubtitle {
path,
language: "de-orig".to_string(),
kind: SubtitleKind::Auto,
format: "vtt".to_string(),
original_language: Some("de".to_string()),
}
}
#[test]
fn register_subtitle_artifacts_dedups_same_blob() {
let (_temp, paths, entry) =
archive_fixture("youtube", "video", Some("https://www.youtube.com/watch?v=x"));
let store_path = &paths.store_path;
let body = "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHallo Welt\n";
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
let first = archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "a.de-orig.vtt", body)]);
assert_eq!(first.len(), 1);
assert!(store_path.join(&first[0].raw_relpath).is_file());
assert_eq!(
register_subtitle_artifacts(&conn, store_path, entry.id, &first, SUBTITLE_ORIGIN_CAPTURE)
.unwrap(),
1
);
// Same bytes fetched again: raw move dedupes, registration skips.
let second = archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "b.de-orig.vtt", body)]);
assert_eq!(second[0].raw_relpath, first[0].raw_relpath);
assert_eq!(
register_subtitle_artifacts(
&conn,
store_path,
entry.id,
&second,
SUBTITLE_ORIGIN_SUMMARY_FETCH
)
.unwrap(),
0
);
let rows =
database::list_entry_artifacts_by_role(&conn, entry.id, SUBTITLE_ARTIFACT_ROLE).unwrap();
assert_eq!(rows.len(), 1);
assert_eq!(rows[0].mime_type.as_deref(), Some("text/vtt"));
assert_eq!(rows[0].relpath, first[0].raw_relpath.to_string_lossy());
assert_eq!(
parse_subtitle_metadata(rows[0].metadata_json.as_deref()),
meta("de-orig", SubtitleKind::Auto, Some("de"))
);
let stored: serde_json::Value =
serde_json::from_str(rows[0].metadata_json.as_deref().unwrap()).unwrap();
assert_eq!(stored["format"], "vtt");
assert_eq!(stored["origin"], SUBTITLE_ORIGIN_CAPTURE);
assert_eq!(usable_subtitle_count(&conn, store_path, entry.id).unwrap(), 1);
assert_eq!(
register_subtitle_artifacts(&conn, store_path, entry.id, &[], SUBTITLE_ORIGIN_CAPTURE)
.unwrap(),
0
);
}
#[test]
fn fetch_subtitles_for_entry_skips_non_youtube_and_non_http_entries() {
// Each of these returns before any yt-dlp process could be spawned.
let (_t1, web_paths, web) = archive_fixture("web", "page", Some("https://example.com/"));
assert_eq!(
fetch_subtitles_for_entry(&web_paths, &web.entry_uid, &[]).unwrap(),
SubtitleFetchOutcome::default()
);
let (_t2, offline_paths, offline) =
archive_fixture("youtube", "video", Some("youtube-test:offline"));
assert_eq!(
fetch_subtitles_for_entry(&offline_paths, &offline.entry_uid, &[]).unwrap(),
SubtitleFetchOutcome::default()
);
let (_t3, no_url_paths, no_url) = archive_fixture("youtube", "video", None);
assert_eq!(
fetch_subtitles_for_entry(&no_url_paths, &no_url.entry_uid, &[]).unwrap(),
SubtitleFetchOutcome::default()
);
assert!(fetch_subtitles_for_entry(&web_paths, "entry_missing", &[]).is_err());
}
#[test]
fn fetch_outcome_reports_original_language_from_existing_artifacts() {
let (_temp, paths, entry) =
archive_fixture("youtube", "video", Some("youtube-test:offline"));
let store_path = &paths.store_path;
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
// An unusable (empty) track still carries the original language.
let archived =
archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "e.de-orig.vtt", "WEBVTT\n")]);
register_subtitle_artifacts(&conn, store_path, entry.id, &archived, SUBTITLE_ORIGIN_CAPTURE)
.unwrap();
assert_eq!(
fetch_subtitles_for_entry(&paths, &entry.entry_uid, &[]).unwrap(),
SubtitleFetchOutcome {
added: 0,
original_language: Some("de".to_string()),
}
);
}
#[test]
fn register_transcript_artifact_writes_engine_metadata_and_dedups() {
let (_temp, paths, entry) =
archive_fixture("youtube", "video", Some("https://www.youtube.com/watch?v=x"));
let store_path = &paths.store_path;
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
let body = "WEBVTT\n\n00:00:00.000 --> 00:00:02.000\nhello world\n";
let staged = |name: &str| {
let mut s = stage_vtt(store_path, name, body);
s.language = "en".to_string();
s.kind = SubtitleKind::Transcribed;
s.original_language = None;
s
};
let first = archive_staged_subtitles(store_path, vec![staged("t1.vtt")]);
assert_eq!(first.len(), 1);
assert_eq!(
register_transcript_artifact(
&conn,
store_path,
entry.id,
&first[0],
"whisper",
"/models/ggml-tiny.bin"
)
.unwrap(),
1
);
let second = archive_staged_subtitles(store_path, vec![staged("t2.vtt")]);
assert_eq!(
register_transcript_artifact(
&conn,
store_path,
entry.id,
&second[0],
"whisper",
"/models/ggml-tiny.bin"
)
.unwrap(),
0
);
let rows =
database::list_entry_artifacts_by_role(&conn, entry.id, SUBTITLE_ARTIFACT_ROLE).unwrap();
assert_eq!(rows.len(), 1);
assert_eq!(rows[0].mime_type.as_deref(), Some("text/vtt"));
let stored: serde_json::Value =
serde_json::from_str(rows[0].metadata_json.as_deref().unwrap()).unwrap();
assert_eq!(stored["kind"], "transcribed");
assert_eq!(stored["origin"], SUBTITLE_ORIGIN_TRANSCRIPTION);
assert_eq!(stored["engine"], "whisper");
assert_eq!(stored["model"], "ggml-tiny.bin");
assert_eq!(stored["language"], "en");
assert_eq!(
parse_subtitle_metadata(rows[0].metadata_json.as_deref()).kind,
SubtitleKind::Transcribed
);
assert_eq!(sanitize_model_name("nvidia/parakeet-tdt-0.6b-v3"), "nvidia/parakeet-tdt-0.6b-v3");
assert_eq!(sanitize_model_name("phonon-2"), "phonon-2");
assert_eq!(sanitize_model_name("C:\\models\\x.bin"), "x.bin");
}
}

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,650 @@
//! On-demand titles for archived X threads.
//!
//! A user asks for a title from the entry rail; the ordered thread text is sent
//! to the selected summary provider with a cheap per-provider model and the
//! result is saved as `Thread about <topic> — @author` (just `Thread about
//! <topic>` when the author is unknown).
//!
//! The title model never inherits the summary model (`ARCHIVR_*_MODEL`). It is
//! resolved as: the admin's instance setting (passed in by the caller; core
//! never reads the auth DB) > `ARCHIVR_ANTHROPIC_TITLE_MODEL` /
//! `ARCHIVR_OPENAI_TITLE_MODEL` / `ARCHIVR_CLAUDE_TITLE_MODEL` /
//! `ARCHIVR_CODEX_TITLE_MODEL` > a built-in small default. Endpoint, key, CLI
//! path and timeout still come from `summarizer::provider_from_env`.
//!
//! The model returns only the topic phrase; the server builds the rest so the
//! title format (prefix and author suffix) is guaranteed regardless of output.
use anyhow::{Context, Result, bail};
use rusqlite::OptionalExtension;
use std::path::Path;
use crate::archive::ArchivePaths;
use crate::database;
use crate::env_config::optional_env;
use crate::summarizer::{self, ProviderConfig};
pub const TITLE_MAX_TOKENS: u32 = 64;
const MAX_TITLE_INPUT_CHARS: usize = 8_000;
const MAX_TOPIC_WORDS: usize = 10;
const MAX_TOPIC_CHARS: usize = 80;
const TITLE_SYSTEM_PROMPT: &str = "You name archived X (Twitter) threads for a personal archive index. Reply with ONLY a short topic phrase of 3 to 8 words that completes the sentence 'Thread about …' (for example: migrating a home server to NixOS). Plain text on one line: no quotes, no markdown, no hashtags, no emoji, no @mentions, no trailing punctuation, and do not repeat the words 'Thread about'.";
const QUOTE_CHARS: &[char] = &['"', '\'', '`', '“', '”', '‘', '’', '«', '»', '*', '_', '#'];
const TRAILING_PUNCT: &[char] = &['.', ',', ';', ':', '!', '?', '…'];
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct ThreadTitleInput {
pub entry_uid: String,
/// Empty when no status JSON names the author.
pub author: String,
pub content: String,
}
pub fn title_model_env(kind: &str) -> Option<&'static str> {
match kind {
"anthropic_http" => Some("ARCHIVR_ANTHROPIC_TITLE_MODEL"),
"openai_compatible" => Some("ARCHIVR_OPENAI_TITLE_MODEL"),
"claude_cli" => Some("ARCHIVR_CLAUDE_TITLE_MODEL"),
"codex_cli" => Some("ARCHIVR_CODEX_TITLE_MODEL"),
_ => None,
}
}
pub fn default_title_model(kind: &str) -> Option<&'static str> {
match kind {
"anthropic_http" => Some("claude-haiku-4-5"),
"openai_compatible" => Some("gpt-4o-mini"),
"claude_cli" => Some("haiku"),
"codex_cli" => Some("gpt-6-luna"),
_ => None,
}
}
pub fn with_title_model(cfg: ProviderConfig, model: String) -> ProviderConfig {
match cfg {
ProviderConfig::AnthropicHttp(mut c) => {
c.model = model;
ProviderConfig::AnthropicHttp(c)
}
ProviderConfig::OpenAiCompatible(mut c) => {
c.model = model;
ProviderConfig::OpenAiCompatible(c)
}
ProviderConfig::ClaudeCli(mut c) => {
c.model = Some(model);
ProviderConfig::ClaudeCli(c)
}
ProviderConfig::CodexCli(mut c) => {
c.model = Some(model);
ProviderConfig::CodexCli(c)
}
}
}
/// Where an effective title model came from.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum TitleModelSource {
Instance,
Env,
Default,
}
impl TitleModelSource {
pub fn as_str(self) -> &'static str {
match self {
Self::Instance => "instance",
Self::Env => "env",
Self::Default => "default",
}
}
}
/// Effective title model for `kind`: non-empty trimmed `instance_override` >
/// non-empty title env var > built-in default. `None` for unknown kinds.
pub fn resolve_title_model(
kind: &str,
instance_override: Option<&str>,
) -> Option<(String, TitleModelSource)> {
let (var, default) = (title_model_env(kind)?, default_title_model(kind)?);
if let Some(m) = instance_override.map(str::trim).filter(|m| !m.is_empty()) {
return Some((m.to_string(), TitleModelSource::Instance));
}
if let Some(m) = optional_env(var).map(|m| m.trim().to_string()).filter(|m| !m.is_empty()) {
return Some((m, TitleModelSource::Env));
}
Some((default.to_string(), TitleModelSource::Default))
}
/// Provider config for title generation: transport settings from the summary
/// env, model from [`resolve_title_model`].
pub fn title_provider_from_env(
kind: &str,
instance_override: Option<&str>,
) -> Result<ProviderConfig> {
// Validates `kind` and keeps the summary path's missing-key messages.
let cfg = summarizer::provider_from_env(kind)?;
let Some((model, _)) = resolve_title_model(kind, instance_override) else {
bail!("unknown summary provider: {kind}");
};
Ok(with_title_model(cfg, model))
}
/// Expected, user-facing failure of [`load_thread_title_input`] (entry is not a
/// thread, or has no archived text). Anything else is an internal error.
#[derive(Debug)]
pub struct ThreadTitleUserError(pub String);
impl std::fmt::Display for ThreadTitleUserError {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(&self.0)
}
}
impl std::error::Error for ThreadTitleUserError {}
/// The [`ThreadTitleUserError`] message carried by `error`, if any.
pub fn thread_title_user_message(error: &anyhow::Error) -> Option<String> {
error.downcast_ref::<ThreadTitleUserError>().map(|m| m.0.clone())
}
/// Loads the thread text and author. `Ok(None)` means the entry does not exist.
pub fn load_thread_title_input(
paths: &ArchivePaths,
entry_uid: &str,
) -> Result<Option<ThreadTitleInput>> {
let conn = database::open_or_initialize(&paths.archive_path)?;
let Some((id, entity_kind, source_metadata_json)) = conn
.query_row(
"SELECT id, entity_kind, source_metadata_json FROM archived_entries WHERE entry_uid = ?1",
[entry_uid],
|row| Ok((row.get::<_, i64>(0)?, row.get::<_, String>(1)?, row.get::<_, String>(2)?)),
)
.optional()?
else {
return Ok(None);
};
if entity_kind != "tweet_thread" {
return Err(anyhow::Error::new(ThreadTitleUserError(format!(
"entry is '{entity_kind}', not an X thread; titles can only be generated for threads"
))));
}
let content = summarizer::artifact_text_content(&conn, &paths.store_path, id, &entity_kind)
.map_err(|e| {
if summarizer::is_unsupported_summary_content_error(&e) {
anyhow::Error::new(ThreadTitleUserError(
"this thread has no archived text to generate a title from".to_string(),
))
} else {
e
}
})?;
let content: String = content.chars().take(MAX_TITLE_INPUT_CHARS).collect();
let root_tweet_id = serde_json::from_str::<serde_json::Value>(&source_metadata_json)
.ok()
.and_then(|v| v["tweet_id"].as_str().map(str::to_string));
let author = thread_author(
&conn,
&paths.store_path,
id,
root_tweet_id.as_deref(),
entry_uid,
)?;
Ok(Some(ThreadTitleInput {
entry_uid: entry_uid.to_string(),
author,
content,
}))
}
/// Author of the root status (`tweet-<source_metadata.tweet_id>.json`, as in
/// capture's `Thread by @…`), else of the first readable status JSON.
fn thread_author(
conn: &rusqlite::Connection,
store_path: &Path,
entry_id: i64,
root_tweet_id: Option<&str>,
entry_uid: &str,
) -> Result<String> {
let root_file = root_tweet_id.map(|id| format!("tweet-{id}.json"));
for role in ["raw_tweet_json", "primary_media"] {
let mut artifacts: Vec<_> = database::list_entry_artifacts_by_role(conn, entry_id, role)?
.into_iter()
.filter(|a| a.relpath.ends_with(".json"))
.collect();
if artifacts.is_empty() {
continue;
}
// Root status first; the rest keep insertion order (stable sort).
if let Some(root_file) = root_file.as_deref() {
artifacts.sort_by_key(|a| {
!(Path::new(&a.relpath).file_name().and_then(|n| n.to_str()) == Some(root_file))
});
}
for artifact in &artifacts {
let abs = store_path.join(&artifact.relpath);
let parsed = std::fs::read_to_string(&abs)
.with_context(|| format!("failed to read {}", abs.display()))
.and_then(|raw| {
serde_json::from_str::<serde_json::Value>(&raw)
.with_context(|| format!("{} is not valid JSON", abs.display()))
});
let json = match parsed {
Ok(json) => json,
Err(e) => {
eprintln!("warn: thread title {entry_uid}: {e:#}");
continue;
}
};
let name = json["author"]["screen_name"]
.as_str()
.map(|s| s.trim().trim_start_matches('@').trim())
.filter(|s| !s.is_empty());
if let Some(name) = name {
return Ok(name.to_string());
}
}
// Only the first role that has JSON artifacts is consulted, matching
// `artifact_text_content`'s legacy `primary_media` fallback.
break;
}
Ok(String::new())
}
pub fn build_title_user_prompt(input: &ThreadTitleInput) -> String {
if input.author.is_empty() {
format!("Thread:\n{}\n", input.content)
} else {
format!("Author: @{}\n\nThread:\n{}\n", input.author, input.content)
}
}
fn strip_prefix_ci<'a>(s: &'a str, prefix: &str) -> Option<&'a str> {
let head = s.get(..prefix.len())?;
head.eq_ignore_ascii_case(prefix).then(|| &s[prefix.len()..])
}
fn trim_trailing_punct(s: &str) -> &str {
s.trim_end_matches(|c: char| TRAILING_PUNCT.contains(&c) || QUOTE_CHARS.contains(&c))
.trim_end()
}
/// Reduces a model reply to a single short topic phrase.
pub fn sanitize_topic(raw: &str) -> Result<String> {
let line = raw
.lines()
.filter(|l| !l.trim().starts_with("```"))
.map(str::trim)
.find(|l| !l.is_empty())
.unwrap_or("");
let line: String = line.chars().filter(|c| !c.is_control()).collect();
let mut s: &str = line.trim();
for prefix in ["title:", "topic:"] {
if let Some(rest) = strip_prefix_ci(s, prefix) {
s = rest;
break;
}
}
loop {
let before = s;
s = s.trim().trim_matches(|c: char| QUOTE_CHARS.contains(&c));
if let Some(rest) = strip_prefix_ci(s, "thread about ") {
s = rest;
}
if let Some(rest) = strip_prefix_ci(s, "about ") {
s = rest;
}
if s == before {
break;
}
}
for sep in [" — @", " – @", " - @"] {
if let Some(idx) = s.find(sep) {
s = &s[..idx];
}
}
let collapsed = s.split_whitespace().collect::<Vec<_>>().join(" ");
let trimmed = trim_trailing_punct(&collapsed);
let mut topic = trimmed
.split(' ')
.filter(|w| !w.is_empty())
.take(MAX_TOPIC_WORDS)
.collect::<Vec<_>>()
.join(" ");
if topic.chars().count() > MAX_TOPIC_CHARS {
let head: String = topic.chars().take(MAX_TOPIC_CHARS).collect();
// `head` is MAX_TOPIC_CHARS chars and the next char exists, so a space
// at the cut point is preserved by checking the following char too.
let next_is_space = topic.chars().nth(MAX_TOPIC_CHARS) == Some(' ');
let cut = if next_is_space {
head.as_str()
} else {
match head.rfind(' ') {
Some(idx) => &head[..idx],
None => head.as_str(),
}
};
topic = trim_trailing_punct(cut).to_string();
}
if topic.is_empty() {
bail!("title provider returned no usable title");
}
Ok(topic)
}
/// `author` is empty when unknown; the ` — @…` suffix is then omitted.
pub fn format_thread_title(topic: &str, author: &str) -> String {
if author.is_empty() {
format!("Thread about {topic}")
} else {
format!("Thread about {topic} — @{author}")
}
}
fn short(s: &str) -> String {
s.chars().take(200).collect()
}
/// Asks the provider for a topic and builds the final title.
pub fn generate_thread_title(cfg: &ProviderConfig, input: &ThreadTitleInput) -> Result<String> {
let out = summarizer::complete_plain(
cfg,
TITLE_SYSTEM_PROMPT,
&build_title_user_prompt(input),
TITLE_MAX_TOKENS,
)
.context("title generation failed")?;
let topic =
sanitize_topic(&out.text).with_context(|| format!("raw reply: {}", short(&out.text)))?;
Ok(format_thread_title(&topic, &input.author))
}
#[cfg(test)]
mod tests {
use super::*;
use crate::summarizer::{CliProviderConfig, HttpProviderConfig};
#[test]
fn sanitize_topic_cleans_model_replies() {
let ok = |raw: &str| sanitize_topic(raw).unwrap();
assert_eq!(ok("\"Rust async runtimes compared.\""), "Rust async runtimes compared");
assert_eq!(ok("Thread about NixOS on a Pi"), "NixOS on a Pi");
assert_eq!(ok("```\nTitle: **Home lab networking**\n```"), "Home lab networking");
assert_eq!(ok("\nFirst line topic\nSecond line"), "First line topic");
assert_eq!(ok("Foo bar — @alice"), "Foo bar");
let twenty = (1..=20).map(|i| format!("w{i}")).collect::<Vec<_>>().join(" ");
assert_eq!(ok(&twenty), "w1 w2 w3 w4 w5 w6 w7 w8 w9 w10");
let long_word = "x".repeat(120);
assert!(ok(&long_word).chars().count() <= MAX_TOPIC_CHARS);
let long_words = vec!["abcdefghijk"; 10].join(" ");
let cut = ok(&long_words);
assert!(cut.chars().count() <= MAX_TOPIC_CHARS);
assert!(!cut.ends_with(' '));
assert!(sanitize_topic(" ").is_err());
assert!(sanitize_topic("\"\"").is_err());
}
#[test]
fn format_thread_title_uses_em_dash_suffix() {
assert_eq!(format_thread_title("x y", "bob"), "Thread about x y — @bob");
assert_eq!(format_thread_title("x y", ""), "Thread about x y");
}
#[test]
fn title_models_cover_all_provider_kinds() {
for kind in summarizer::PROVIDER_KINDS {
assert!(title_model_env(kind).is_some(), "{kind}");
assert!(default_title_model(kind).is_some(), "{kind}");
}
assert_eq!(title_model_env("claude_cli"), Some("ARCHIVR_CLAUDE_TITLE_MODEL"));
assert_eq!(default_title_model("codex_cli"), Some("gpt-6-luna"));
assert_eq!(default_title_model("anthropic_http"), Some("claude-haiku-4-5"));
assert_eq!(title_model_env("gemini"), None);
assert_eq!(default_title_model("gemini"), None);
}
#[test]
fn with_title_model_sets_model_on_every_variant() {
let http = HttpProviderConfig {
endpoint: "https://example.invalid".into(),
api_key: "k".into(),
model: "big".into(),
timeout_secs: 1,
};
let cli = CliProviderConfig {
executable: "claude".into(),
model: None,
timeout_secs: 1,
};
let m = || "small".to_string();
match with_title_model(ProviderConfig::AnthropicHttp(http.clone()), m()) {
ProviderConfig::AnthropicHttp(c) => assert_eq!(c.model, "small"),
other => panic!("{other:?}"),
}
match with_title_model(ProviderConfig::OpenAiCompatible(http), m()) {
ProviderConfig::OpenAiCompatible(c) => assert_eq!(c.model, "small"),
other => panic!("{other:?}"),
}
match with_title_model(ProviderConfig::ClaudeCli(cli.clone()), m()) {
ProviderConfig::ClaudeCli(c) => assert_eq!(c.model.as_deref(), Some("small")),
other => panic!("{other:?}"),
}
match with_title_model(ProviderConfig::CodexCli(cli), m()) {
ProviderConfig::CodexCli(c) => assert_eq!(c.model.as_deref(), Some("small")),
other => panic!("{other:?}"),
}
}
/// Archive with one entry of `entity_kind` and the given raw tweet JSON files.
fn fixture(
entity_kind: &str,
tweets: &[(&str, serde_json::Value)],
) -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
let temp = tempfile::tempdir().unwrap();
let paths = crate::archive::initialize_archive(
temp.path(),
&temp.path().join("store"),
"Test archive",
false,
)
.unwrap();
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
let user_id = database::ensure_default_user(&conn).unwrap();
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
let source_id = database::upsert_source_identity(
&conn,
"x",
entity_kind,
Some("9001"),
Some("https://x.com/alice/status/9001"),
"x:thread:9001",
)
.unwrap();
let entry = database::create_archived_entry(
&conn,
&database::NewEntry {
source_identity_id: source_id,
archive_run_id: run.id,
parent_entry_id: None,
root_entry_id: None,
created_by_user_id: user_id,
owned_by_user_id: user_id,
source_kind: "x".to_string(),
entity_kind: entity_kind.to_string(),
title: Some("Thread by @alice".to_string()),
visibility: "private".to_string(),
representation_kind: entity_kind.to_string(),
source_metadata_json: r#"{"tweet_id":"9001"}"#.to_string(),
display_metadata_json: None,
},
)
.unwrap();
std::fs::create_dir_all(paths.store_path.join("raw_tweets")).unwrap();
for (relpath, body) in tweets {
std::fs::write(paths.store_path.join(relpath), body.to_string()).unwrap();
database::add_entry_artifact(
&conn,
&database::NewArtifact {
entry_id: entry.id,
artifact_role: "raw_tweet_json".to_string(),
storage_area: "raw_tweets".to_string(),
relpath: relpath.to_string(),
blob_id: None,
logical_path: None,
metadata_json: None,
},
)
.unwrap();
}
(temp, paths, entry)
}
fn alice_thread() -> Vec<(&'static str, serde_json::Value)> {
vec![
(
"raw_tweets/tweet-9001.json",
serde_json::json!({
"full_text": "1/ Comparing Rust async runtimes.",
"author": { "screen_name": "@alice" }
}),
),
(
"raw_tweets/tweet-9002.json",
serde_json::json!({
"full_text": "2/ Tokio wins on ecosystem.",
"author": { "screen_name": "alice" }
}),
),
]
}
#[test]
fn load_thread_title_input_reads_thread_text_and_author() {
let (_temp, paths, entry) = fixture("tweet_thread", &alice_thread());
let input = load_thread_title_input(&paths, &entry.entry_uid)
.unwrap()
.unwrap();
assert_eq!(input.entry_uid, entry.entry_uid);
assert_eq!(input.author, "alice");
assert!(
input
.content
.contains("1/ Comparing Rust async runtimes.\n\n---\n\n2/ Tokio wins on ecosystem."),
"{}",
input.content
);
assert!(!input.content.contains("Thread by @alice"));
}
#[test]
fn load_thread_title_input_prefers_root_status_author() {
let tweets = vec![
(
"raw_tweets/tweet-8000.json",
serde_json::json!({ "full_text": "quoted", "author": { "screen_name": "bob" } }),
),
(
"raw_tweets/tweet-9001.json",
serde_json::json!({ "full_text": "root", "author": { "screen_name": "alice" } }),
),
];
let (_temp, paths, entry) = fixture("tweet_thread", &tweets);
let input = load_thread_title_input(&paths, &entry.entry_uid)
.unwrap()
.unwrap();
assert_eq!(input.author, "alice");
}
#[test]
fn load_thread_title_input_rejects_non_threads_and_empty_threads() {
let (_temp, paths, entry) = fixture("page", &[]);
let err = load_thread_title_input(&paths, &entry.entry_uid).unwrap_err();
assert!(format!("{err:#}").contains("not an X thread"), "{err:#}");
assert!(thread_title_user_message(&err).is_some());
assert!(load_thread_title_input(&paths, "no-such-uid").unwrap().is_none());
let empty = vec![(
"raw_tweets/tweet-9001.json",
serde_json::json!({ "full_text": "", "author": { "screen_name": "alice" } }),
)];
let (_temp2, paths2, entry2) = fixture("tweet_thread", &empty);
let err = load_thread_title_input(&paths2, &entry2.entry_uid).unwrap_err();
assert!(format!("{err:#}").contains("no archived text"), "{err:#}");
assert!(thread_title_user_message(&err).is_some());
}
#[cfg(unix)]
#[test]
fn generate_thread_title_runs_claude_cli_with_title_model() {
let dir = tempfile::tempdir().unwrap();
let script = dir.path().join("fake-claude");
crate::downloader::write_script(
&script,
"#!/bin/sh\ncat >/dev/null\ncase \" $* \" in *\" --model haiku \"*) ;; *) echo \"bad args: $*\" >&2; exit 3;; esac\nprintf '%s\\n' '\"Rust async runtimes compared.\"'\n",
);
let cfg = ProviderConfig::ClaudeCli(CliProviderConfig {
executable: script,
model: Some("haiku".into()),
timeout_secs: 30,
});
let input = ThreadTitleInput {
entry_uid: "uid".into(),
author: "alice".into(),
content: "1/ Comparing Rust async runtimes.".into(),
};
// Parallel tests forking while the script fd was open can briefly make
// exec fail with ETXTBSY (rust-lang/rust#114554); retry that case only.
let mut attempt = 0;
let title = loop {
match generate_thread_title(&cfg, &input) {
Err(e) if attempt < 20 && format!("{e:#}").contains("Text file busy") => {
attempt += 1;
std::thread::sleep(std::time::Duration::from_millis(50));
}
other => break other.unwrap(),
}
};
assert_eq!(title, "Thread about Rust async runtimes compared — @alice");
}
#[test]
fn resolve_title_model_prefers_instance_then_env_then_default() {
// Only this test touches the codex title env var; restored below.
let var = "ARCHIVR_CODEX_TITLE_MODEL";
let previous = std::env::var_os(var);
unsafe { std::env::remove_var(var) };
assert_eq!(
resolve_title_model("codex_cli", None),
Some(("gpt-6-luna".into(), TitleModelSource::Default))
);
assert_eq!(
resolve_title_model("codex_cli", Some(" ")),
Some(("gpt-6-luna".into(), TitleModelSource::Default))
);
unsafe { std::env::set_var(var, " env-model ") };
assert_eq!(
resolve_title_model("codex_cli", None),
Some(("env-model".into(), TitleModelSource::Env))
);
assert_eq!(
resolve_title_model("codex_cli", Some(" inst-model ")),
Some(("inst-model".into(), TitleModelSource::Instance))
);
unsafe {
match previous {
Some(v) => std::env::set_var(var, v),
None => std::env::remove_var(var),
}
}
assert_eq!(resolve_title_model("gemini", Some("x")), None);
assert_eq!(TitleModelSource::Env.as_str(), "env");
}
}

File diff suppressed because it is too large Load diff