mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 21:03:17 +02:00
Add YouTube subtitles, local transcription, self-updating yt-dlp/Deno, X Article and thread titles
- Capture YouTube subtitles by default (opt-out in UI, API, CLI --no-subtitles) - Summarize YouTube videos from subtitles; fetch on demand, then local transcription, then error - Local transcription fallback: Whisper, Parakeet, Phonon-2 (English only) - Runtime-resolved, self-updating yt-dlp and Deno JS runtime (fixes YouTube 403s) - Settings > Instance > yt-dlp: status and in-app update without restart - X Article titles from article.title, with idempotent startup backfill - Thread title generation (single and bulk) with per-provider cheap models - Per-provider title model settings in Settings > Instance - Docs, mental model, AGENTS.md and transcription spec updated
This commit is contained in:
parent
4f3b2968b6
commit
253f779216
35 changed files with 11377 additions and 583 deletions
|
|
@ -15,6 +15,10 @@ sha3.workspace = true
|
|||
uuid.workspace = true
|
||||
reqwest = { workspace = true }
|
||||
base64.workspace = true
|
||||
zip.workspace = true
|
||||
|
||||
[target.'cfg(unix)'.dependencies]
|
||||
libc.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
use crate::{
|
||||
archive::{self, ArchivePaths},
|
||||
database, downloader,
|
||||
database, downloader, subtitles,
|
||||
twitter::parse_tweet_id,
|
||||
};
|
||||
use anyhow::{Context, Result};
|
||||
|
|
@ -83,7 +83,7 @@ impl PlatformMetadata {
|
|||
|
||||
/// Configuration passed to `perform_capture` to supply per-instance settings
|
||||
/// that live outside the archive (e.g. cookies stored in the auth DB).
|
||||
#[derive(Debug, Clone, Default)]
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct CaptureConfig {
|
||||
pub cookie_rules: Vec<database::CookieRule>,
|
||||
/// Override for uBlock Origin Lite during WebPage captures.
|
||||
|
|
@ -108,6 +108,38 @@ pub struct CaptureConfig {
|
|||
/// When true, skip playlist items whose URL is already archived as a child
|
||||
/// of any container entry with the same canonical playlist URL.
|
||||
pub sync: bool,
|
||||
/// Download subtitles for YouTube videos (manual preferred, auto fallback). Default true.
|
||||
pub download_subtitles: bool,
|
||||
}
|
||||
|
||||
impl Default for CaptureConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
cookie_rules: Vec::new(),
|
||||
ublock_enabled: None,
|
||||
cookie_ext_enabled: None,
|
||||
reader_mode: false,
|
||||
modal_closer_enabled: None,
|
||||
via_freedium: false,
|
||||
per_item_quality: HashMap::new(),
|
||||
sync: false,
|
||||
download_subtitles: true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Plan a subtitle request from the yt-dlp metadata probe. Only YouTube videos
|
||||
/// (not YouTube Music / audio) get subtitles, and only when enabled.
|
||||
fn subtitle_request_for(
|
||||
source: Source,
|
||||
config: &CaptureConfig,
|
||||
metadata_json: Option<&str>,
|
||||
) -> Option<downloader::ytdlp::SubtitleRequest> {
|
||||
if config.download_subtitles && source == Source::YouTubeVideo {
|
||||
downloader::ytdlp::plan_subtitle_request(metadata_json)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolves which cookies apply to `url` by evaluating all rules in ordinal order.
|
||||
|
|
@ -238,12 +270,17 @@ fn generate_entry_title(source: Source, meta: &PlatformMetadata) -> String {
|
|||
.unwrap_or_else(|| "Spotify Content".to_string()),
|
||||
Source::X => format!("X Media by {}", meta.author.as_deref().unwrap_or("unknown")),
|
||||
Source::Tweet => {
|
||||
let excerpt = meta
|
||||
.caption_excerpt()
|
||||
let headline = meta
|
||||
.title
|
||||
.as_deref()
|
||||
.map(str::trim)
|
||||
.filter(|t| !t.is_empty())
|
||||
.map(str::to_string)
|
||||
.or_else(|| meta.caption_excerpt())
|
||||
.unwrap_or_else(|| "Tweet".to_string());
|
||||
format!(
|
||||
"{} \u{2014} @{}",
|
||||
excerpt,
|
||||
headline,
|
||||
meta.author.as_deref().unwrap_or("unknown")
|
||||
)
|
||||
}
|
||||
|
|
@ -911,6 +948,7 @@ fn record_container_entry(
|
|||
}
|
||||
|
||||
/// Extracts PlatformMetadata from a tweet JSON string.
|
||||
/// `title` is the X Article title when the status is an Article.
|
||||
/// Returns Default on any parse failure.
|
||||
fn tweet_metadata_from_json(json_str: &str) -> PlatformMetadata {
|
||||
let Ok(v) = serde_json::from_str::<serde_json::Value>(json_str) else {
|
||||
|
|
@ -930,8 +968,16 @@ fn tweet_metadata_from_json(json_str: &str) -> PlatformMetadata {
|
|||
.map(|s| s.trim().to_string())
|
||||
.filter(|s| !s.is_empty());
|
||||
|
||||
let article_title = v
|
||||
.get("article")
|
||||
.and_then(|a| a.get("title"))
|
||||
.and_then(|t| t.as_str())
|
||||
.map(|s| s.trim().to_string())
|
||||
.filter(|s| !s.is_empty());
|
||||
|
||||
PlatformMetadata {
|
||||
author: screen_name,
|
||||
title: article_title,
|
||||
caption: full_text,
|
||||
..Default::default()
|
||||
}
|
||||
|
|
@ -1050,6 +1096,69 @@ fn record_tweet_entry(
|
|||
Ok(entry)
|
||||
}
|
||||
|
||||
/// Rewrites legacy bare-link titles of X Article tweet entries to
|
||||
/// "<article title> — @handle". Idempotent and safe to run on every start:
|
||||
/// a row is only rewritten while its title still byte-equals the legacy title
|
||||
/// recomputed from its raw JSON (so user renames are never touched), and the
|
||||
/// write is compare-and-set. Rows with missing/unreadable raw JSON are skipped
|
||||
/// with a `warn:` line and retried next run. Returns the number of rows changed.
|
||||
pub fn backfill_x_article_titles(paths: &archive::ArchivePaths) -> Result<usize> {
|
||||
let conn = database::open_or_initialize(&paths.archive_path)?;
|
||||
let mut count = 0;
|
||||
for c in database::list_bare_link_tweet_titles(&conn)? {
|
||||
let Ok(source_meta) = serde_json::from_str::<serde_json::Value>(&c.source_metadata_json)
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let Some(tweet_id) = source_meta.get("tweet_id").and_then(|v| v.as_str()) else {
|
||||
continue;
|
||||
};
|
||||
if tweet_id.is_empty() || !tweet_id.bytes().all(|b| b.is_ascii_digit()) {
|
||||
continue;
|
||||
}
|
||||
let json_path = paths
|
||||
.store_path
|
||||
.join("raw_tweets")
|
||||
.join(format!("tweet-{tweet_id}.json"));
|
||||
let json = match fs::read_to_string(&json_path) {
|
||||
Ok(json) => json,
|
||||
Err(e) => {
|
||||
eprintln!("warn: X Article title backfill: skipping entry {}: {e}", c.id);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let meta = tweet_metadata_from_json(&json);
|
||||
if meta.title.is_none() {
|
||||
continue;
|
||||
}
|
||||
let bare_link = meta.caption.as_deref().map(str::trim).is_some_and(|t| {
|
||||
!t.chars().any(char::is_whitespace)
|
||||
&& (t.starts_with("https://") || t.starts_with("http://"))
|
||||
});
|
||||
if !bare_link {
|
||||
continue;
|
||||
}
|
||||
let legacy = generate_entry_title(
|
||||
Source::Tweet,
|
||||
&PlatformMetadata {
|
||||
title: None,
|
||||
..meta.clone()
|
||||
},
|
||||
);
|
||||
if c.title != legacy {
|
||||
continue;
|
||||
}
|
||||
let new_title = generate_entry_title(Source::Tweet, &meta);
|
||||
if new_title == legacy {
|
||||
continue;
|
||||
}
|
||||
if database::replace_entry_title_if_unchanged(&conn, c.id, &legacy, &new_title)? {
|
||||
count += 1;
|
||||
}
|
||||
}
|
||||
Ok(count)
|
||||
}
|
||||
|
||||
/// Trusted image MIME types emitted by the X downloader's media paths.
|
||||
///
|
||||
/// Tweet JSON has no MIME field for ordinary downloaded media. Restricting this
|
||||
|
|
@ -1375,15 +1484,26 @@ pub fn perform_capture(
|
|||
.or(child_quality)
|
||||
};
|
||||
|
||||
let child_source = match source {
|
||||
Source::SpotifyAlbum | Source::SpotifyPlaylist => Source::SpotifyTrack,
|
||||
_ if is_audio => Source::YouTubeMusicTrack,
|
||||
_ => Source::YouTubeVideo,
|
||||
};
|
||||
let child_subtitle_request =
|
||||
subtitle_request_for(child_source, config, child_meta_json.as_deref());
|
||||
|
||||
// Download the media.
|
||||
match downloader::ytdlp::download(
|
||||
playlist_item.url.clone(),
|
||||
store_path,
|
||||
&child_timestamp,
|
||||
effective_child_quality,
|
||||
child_subtitle_request.as_ref(),
|
||||
&cookies,
|
||||
) {
|
||||
Ok((hash, file_extension)) => {
|
||||
Ok(dl) => {
|
||||
let hash = dl.hash;
|
||||
let file_extension = dl.extension;
|
||||
let temp_file = store_path
|
||||
.join("temp")
|
||||
.join(&child_timestamp)
|
||||
|
|
@ -1411,13 +1531,10 @@ pub fn perform_capture(
|
|||
continue;
|
||||
}
|
||||
}
|
||||
let archived_subtitles =
|
||||
subtitles::archive_staged_subtitles(store_path, dl.subtitles);
|
||||
let _ = fs::remove_dir_all(store_path.join("temp").join(&child_timestamp));
|
||||
|
||||
let child_source = match source {
|
||||
Source::SpotifyAlbum | Source::SpotifyPlaylist => Source::SpotifyTrack,
|
||||
_ if is_audio => Source::YouTubeMusicTrack,
|
||||
_ => Source::YouTubeVideo,
|
||||
};
|
||||
match record_media_entry(
|
||||
&conn,
|
||||
store_path,
|
||||
|
|
@ -1435,6 +1552,18 @@ pub fn perform_capture(
|
|||
Some(container_id),
|
||||
) {
|
||||
Ok(child_entry) => {
|
||||
if let Err(e) = subtitles::register_subtitle_artifacts(
|
||||
&conn,
|
||||
store_path,
|
||||
child_entry.id,
|
||||
&archived_subtitles,
|
||||
subtitles::SUBTITLE_ORIGIN_CAPTURE,
|
||||
) {
|
||||
eprintln!(
|
||||
"warn: register subtitles for {}: {e:#}",
|
||||
playlist_item.url
|
||||
);
|
||||
}
|
||||
let _ = database::refresh_entry_cached_bytes(&conn, child_entry.id);
|
||||
}
|
||||
Err(e) => {
|
||||
|
|
@ -1829,7 +1958,9 @@ pub fn perform_capture(
|
|||
_ => None,
|
||||
};
|
||||
|
||||
let (hash, file_extension) = match source {
|
||||
let subtitle_request = subtitle_request_for(source, config, ytdlp_metadata_json.as_deref());
|
||||
|
||||
let (hash, file_extension, staged_subtitles) = match source {
|
||||
Source::YouTubeVideo
|
||||
| Source::X
|
||||
| Source::Instagram
|
||||
|
|
@ -1842,10 +1973,12 @@ pub fn perform_capture(
|
|||
store_path,
|
||||
×tamp,
|
||||
quality,
|
||||
subtitle_request.as_ref(),
|
||||
&cookies,
|
||||
) {
|
||||
Ok(result) => result,
|
||||
Ok(d) => (d.hash, d.extension, d.subtitles),
|
||||
Err(e) => {
|
||||
let _ = fs::remove_dir_all(store_path.join("temp").join(×tamp));
|
||||
return Err(fail_run(
|
||||
&conn,
|
||||
&run,
|
||||
|
|
@ -1862,10 +1995,12 @@ pub fn perform_capture(
|
|||
store_path,
|
||||
×tamp,
|
||||
Some("audio"),
|
||||
None,
|
||||
&cookies,
|
||||
) {
|
||||
Ok(result) => result,
|
||||
Ok(d) => (d.hash, d.extension, d.subtitles),
|
||||
Err(e) => {
|
||||
let _ = fs::remove_dir_all(store_path.join("temp").join(×tamp));
|
||||
return Err(fail_run(
|
||||
&conn,
|
||||
&run,
|
||||
|
|
@ -1876,7 +2011,7 @@ pub fn perform_capture(
|
|||
}
|
||||
}
|
||||
Source::Local => match downloader::local::save(path.clone(), store_path, ×tamp) {
|
||||
Ok(h) => (h, local_file_extension(&path)),
|
||||
Ok(h) => (h, local_file_extension(&path), Vec::new()),
|
||||
Err(e) => {
|
||||
return Err(fail_run(
|
||||
&conn,
|
||||
|
|
@ -1893,9 +2028,17 @@ pub fn perform_capture(
|
|||
.join("temp")
|
||||
.join(×tamp)
|
||||
.join(format!("{timestamp}{file_extension}"));
|
||||
let byte_size = fs::metadata(&temp_file)
|
||||
.with_context(|| format!("failed to stat staged file {}", temp_file.display()))?
|
||||
.len() as i64;
|
||||
let byte_size = match fs::metadata(&temp_file) {
|
||||
Ok(meta) => meta.len() as i64,
|
||||
Err(e) => {
|
||||
let _ = fs::remove_dir_all(store_path.join("temp").join(×tamp));
|
||||
return Err(anyhow::Error::new(e)
|
||||
.context(format!("failed to stat staged file {}", temp_file.display())));
|
||||
}
|
||||
};
|
||||
|
||||
// Archive subtitle sidecars before the temp dir is removed below.
|
||||
let archived_subtitles = subtitles::archive_staged_subtitles(store_path, staged_subtitles);
|
||||
|
||||
let hash_exists = hash_exists(&hash, &file_extension, store_path)?;
|
||||
|
||||
|
|
@ -1942,6 +2085,15 @@ pub fn perform_capture(
|
|||
None,
|
||||
None,
|
||||
)?;
|
||||
if let Err(e) = subtitles::register_subtitle_artifacts(
|
||||
&conn,
|
||||
store_path,
|
||||
media_entry.id,
|
||||
&archived_subtitles,
|
||||
subtitles::SUBTITLE_ORIGIN_CAPTURE,
|
||||
) {
|
||||
eprintln!("warn: register subtitles for {path}: {e:#}");
|
||||
}
|
||||
database::refresh_entry_cached_bytes(&conn, media_entry.id)?;
|
||||
database::finish_archive_run(&conn, run.id)?;
|
||||
|
||||
|
|
@ -3452,6 +3604,163 @@ mod tests {
|
|||
meta.caption,
|
||||
Some("Hello Rust world, this is a test tweet".to_string())
|
||||
);
|
||||
assert_eq!(meta.title, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tweet_prefers_article_title() {
|
||||
let m = meta(
|
||||
Some("undefinedKi"),
|
||||
Some("Why Boring Wins"),
|
||||
Some("https://t.co/sDrzjUhCzy"),
|
||||
None,
|
||||
None,
|
||||
);
|
||||
assert_eq!(
|
||||
generate_entry_title(Source::Tweet, &m),
|
||||
"Why Boring Wins \u{2014} @undefinedKi"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tweet_blank_article_title_falls_back_to_caption() {
|
||||
let m = meta(Some("alice"), Some(" "), Some("Hello"), None, None);
|
||||
assert_eq!(
|
||||
generate_entry_title(Source::Tweet, &m),
|
||||
"Hello \u{2014} @alice"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tweet_title_extracted_from_x_article_json() {
|
||||
let json = r#"{"full_text":"https://t.co/sDrzjUhCzy","author":{"screen_name":"undefinedKi"},"is_article":true,"article":{"title":" Why Boring Wins ","plain_text":"body"}}"#;
|
||||
let meta = tweet_metadata_from_json(json);
|
||||
assert_eq!(meta.title.as_deref(), Some("Why Boring Wins"));
|
||||
assert_eq!(meta.caption.as_deref(), Some("https://t.co/sDrzjUhCzy"));
|
||||
assert_eq!(
|
||||
generate_entry_title(Source::Tweet, &meta),
|
||||
"Why Boring Wins \u{2014} @undefinedKi"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
mod x_article_backfill_tests {
|
||||
use super::*;
|
||||
|
||||
const ARTICLE_JSON: &str = r#"{"full_text":"https://t.co/sDrzjUhCzy","author":{"screen_name":"undefinedKi"},"is_article":true,"article":{"title":"Why Boring Wins","plain_text":"body"}}"#;
|
||||
const PLAIN_LINK_JSON: &str =
|
||||
r#"{"full_text":"https://t.co/sDrzjUhCzy","author":{"screen_name":"undefinedKi"}}"#;
|
||||
const LEGACY: &str = "https://t.co/sDrzjUhCzy \u{2014} @undefinedKi";
|
||||
|
||||
struct Fixture {
|
||||
_temp: tempfile::TempDir,
|
||||
paths: archive::ArchivePaths,
|
||||
conn: rusqlite::Connection,
|
||||
entry: database::ArchivedEntry,
|
||||
}
|
||||
|
||||
fn fixture(tweet_json: &str) -> Fixture {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let paths = archive::initialize_archive(
|
||||
temp.path(),
|
||||
&temp.path().join("store"),
|
||||
"X Article backfill test",
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
fs::write(
|
||||
paths.store_path.join("raw_tweets").join("tweet-555.json"),
|
||||
tweet_json,
|
||||
)
|
||||
.unwrap();
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
let user_id = database::ensure_default_user(&conn).unwrap();
|
||||
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
|
||||
let item = database::create_archive_run_item(
|
||||
&conn, run.id, None, 0, "tweet:555", None, "x", "tweet",
|
||||
)
|
||||
.unwrap();
|
||||
let entry = record_tweet_entry(
|
||||
&conn,
|
||||
&paths.store_path,
|
||||
user_id,
|
||||
&run,
|
||||
&item,
|
||||
"tweet:555",
|
||||
Source::Tweet,
|
||||
"555",
|
||||
&["raw_tweets/tweet-555.json".to_string()],
|
||||
)
|
||||
.unwrap();
|
||||
Fixture {
|
||||
_temp: temp,
|
||||
paths,
|
||||
conn,
|
||||
entry,
|
||||
}
|
||||
}
|
||||
|
||||
impl Fixture {
|
||||
fn set_title(&self, title: &str) {
|
||||
database::update_entry_title(&self.conn, &self.entry.entry_uid, Some(title))
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
fn title(&self) -> String {
|
||||
self.conn
|
||||
.query_row(
|
||||
"SELECT title FROM archived_entries WHERE id = ?1",
|
||||
[self.entry.id],
|
||||
|row| row.get(0),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn new_capture_of_article_uses_article_title() {
|
||||
let f = fixture(ARTICLE_JSON);
|
||||
assert_eq!(f.title(), "Why Boring Wins \u{2014} @undefinedKi");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn backfill_retitles_legacy_bare_link_article_and_is_idempotent() {
|
||||
let f = fixture(ARTICLE_JSON);
|
||||
f.set_title(LEGACY);
|
||||
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 1);
|
||||
assert_eq!(f.title(), "Why Boring Wins \u{2014} @undefinedKi");
|
||||
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
|
||||
assert_eq!(f.title(), "Why Boring Wins \u{2014} @undefinedKi");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn backfill_preserves_user_edited_title() {
|
||||
let f = fixture(ARTICLE_JSON);
|
||||
f.set_title("My notes");
|
||||
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
|
||||
assert_eq!(f.title(), "My notes");
|
||||
|
||||
f.set_title("https://t.co/sDrzjUhCzy \u{2014} my pick");
|
||||
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
|
||||
assert_eq!(f.title(), "https://t.co/sDrzjUhCzy \u{2014} my pick");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn backfill_ignores_bare_link_tweet_without_article() {
|
||||
let f = fixture(PLAIN_LINK_JSON);
|
||||
assert_eq!(f.title(), LEGACY);
|
||||
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
|
||||
assert_eq!(f.title(), LEGACY);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn backfill_skips_missing_raw_json() {
|
||||
let f = fixture(ARTICLE_JSON);
|
||||
fs::remove_file(f.paths.store_path.join("raw_tweets").join("tweet-555.json"))
|
||||
.unwrap();
|
||||
f.set_title(LEGACY);
|
||||
assert_eq!(backfill_x_article_titles(&f.paths).unwrap(), 0);
|
||||
assert_eq!(f.title(), LEGACY);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -3608,4 +3917,36 @@ mod tests {
|
|||
assert!(!is_freedium_supported_url("https://notmedium.com/article"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn capture_config_default_downloads_subtitles() {
|
||||
let config = CaptureConfig::default();
|
||||
assert!(config.download_subtitles);
|
||||
assert!(config.cookie_rules.is_empty());
|
||||
assert!(!config.reader_mode);
|
||||
assert!(!config.via_freedium);
|
||||
assert!(!config.sync);
|
||||
assert!(config.per_item_quality.is_empty());
|
||||
assert_eq!(config.ublock_enabled, None);
|
||||
assert_eq!(config.cookie_ext_enabled, None);
|
||||
assert_eq!(config.modal_closer_enabled, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subtitle_request_only_for_youtube_video_when_enabled() {
|
||||
let meta = r#"{"language":"en","subtitles":{"en":[{"ext":"vtt"}]},"automatic_captions":{}}"#;
|
||||
let enabled = CaptureConfig::default();
|
||||
let disabled = CaptureConfig {
|
||||
download_subtitles: false,
|
||||
..CaptureConfig::default()
|
||||
};
|
||||
assert!(subtitle_request_for(Source::YouTubeVideo, &enabled, Some(meta)).is_some());
|
||||
assert!(subtitle_request_for(Source::YouTubeVideo, &disabled, Some(meta)).is_none());
|
||||
for source in [Source::TikTok, Source::YouTubeMusicTrack, Source::X] {
|
||||
assert!(
|
||||
subtitle_request_for(source, &enabled, Some(meta)).is_none(),
|
||||
"{source:?} must not request subtitles"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -176,12 +176,47 @@ pub struct InstanceSettings {
|
|||
/// A caller may reorder iff `role_bits & reorder_children_role_bits != 0`.
|
||||
/// Only the Owner may change it. Never contains the Guest bit.
|
||||
pub reorder_children_role_bits: u32,
|
||||
/// Admin overrides for the thread-title model per summary provider kind.
|
||||
/// `None` = fall back to `ARCHIVR_*_TITLE_MODEL` env, then built-in default.
|
||||
pub title_model_anthropic_http: Option<String>,
|
||||
pub title_model_openai_compatible: Option<String>,
|
||||
pub title_model_claude_cli: Option<String>,
|
||||
pub title_model_codex_cli: Option<String>,
|
||||
}
|
||||
|
||||
impl InstanceSettings {
|
||||
pub fn can_reorder_children(&self, role_bits: u32) -> bool {
|
||||
role_bits & self.reorder_children_role_bits != 0
|
||||
}
|
||||
|
||||
/// Instance title-model override for a provider kind (trimmed, non-empty).
|
||||
pub fn title_model_override(&self, kind: &str) -> Option<&str> {
|
||||
self.title_model_slot(kind)?
|
||||
.as_deref()
|
||||
.map(str::trim)
|
||||
.filter(|m| !m.is_empty())
|
||||
}
|
||||
|
||||
/// Mutable column for a provider kind's title model; `None` for unknown kinds.
|
||||
pub fn title_model_slot_mut(&mut self, kind: &str) -> Option<&mut Option<String>> {
|
||||
match kind {
|
||||
"anthropic_http" => Some(&mut self.title_model_anthropic_http),
|
||||
"openai_compatible" => Some(&mut self.title_model_openai_compatible),
|
||||
"claude_cli" => Some(&mut self.title_model_claude_cli),
|
||||
"codex_cli" => Some(&mut self.title_model_codex_cli),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn title_model_slot(&self, kind: &str) -> Option<&Option<String>> {
|
||||
match kind {
|
||||
"anthropic_http" => Some(&self.title_model_anthropic_http),
|
||||
"openai_compatible" => Some(&self.title_model_openai_compatible),
|
||||
"claude_cli" => Some(&self.title_model_claude_cli),
|
||||
"codex_cli" => Some(&self.title_model_codex_cli),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)]
|
||||
|
|
@ -666,7 +701,11 @@ pub fn initialize_auth_schema(conn: &Connection) -> Result<()> {
|
|||
ublock_enabled INTEGER NOT NULL DEFAULT 1 CHECK (ublock_enabled IN (0, 1)),
|
||||
cookie_ext_enabled INTEGER NOT NULL DEFAULT 1 CHECK (cookie_ext_enabled IN (0, 1)),
|
||||
modal_closer_enabled INTEGER NOT NULL DEFAULT 1 CHECK (modal_closer_enabled IN (0, 1)),
|
||||
reorder_children_role_bits INTEGER NOT NULL DEFAULT 12
|
||||
reorder_children_role_bits INTEGER NOT NULL DEFAULT 12,
|
||||
title_model_anthropic_http TEXT,
|
||||
title_model_openai_compatible TEXT,
|
||||
title_model_claude_cli TEXT,
|
||||
title_model_codex_cli TEXT
|
||||
);
|
||||
|
||||
INSERT OR IGNORE INTO instance_settings
|
||||
|
|
@ -726,6 +765,18 @@ pub fn initialize_auth_schema(conn: &Connection) -> Result<()> {
|
|||
"ALTER TABLE instance_settings ADD COLUMN reorder_children_role_bits INTEGER NOT NULL DEFAULT 12",
|
||||
[],
|
||||
);
|
||||
// Add nullable per-provider thread-title model overrides (idempotent migration)
|
||||
for column in [
|
||||
"title_model_anthropic_http",
|
||||
"title_model_openai_compatible",
|
||||
"title_model_claude_cli",
|
||||
"title_model_codex_cli",
|
||||
] {
|
||||
let _ = conn.execute(
|
||||
&format!("ALTER TABLE instance_settings ADD COLUMN {column} TEXT"),
|
||||
[],
|
||||
);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
|
@ -952,7 +1003,9 @@ pub fn get_instance_settings(conn: &Connection) -> Result<InstanceSettings> {
|
|||
COALESCE(ublock_enabled, 1),
|
||||
COALESCE(cookie_ext_enabled, 1),
|
||||
COALESCE(modal_closer_enabled, 1),
|
||||
COALESCE(reorder_children_role_bits, 12)
|
||||
COALESCE(reorder_children_role_bits, 12),
|
||||
title_model_anthropic_http, title_model_openai_compatible,
|
||||
title_model_claude_cli, title_model_codex_cli
|
||||
FROM instance_settings WHERE id = 1",
|
||||
[],
|
||||
|row| {
|
||||
|
|
@ -965,6 +1018,10 @@ pub fn get_instance_settings(conn: &Connection) -> Result<InstanceSettings> {
|
|||
cookie_ext_enabled: row.get::<_, i64>(5)? != 0,
|
||||
modal_closer_enabled: row.get::<_, i64>(6)? != 0,
|
||||
reorder_children_role_bits: row.get::<_, i64>(7)? as u32,
|
||||
title_model_anthropic_http: row.get(8)?,
|
||||
title_model_openai_compatible: row.get(9)?,
|
||||
title_model_claude_cli: row.get(10)?,
|
||||
title_model_codex_cli: row.get(11)?,
|
||||
})
|
||||
},
|
||||
)
|
||||
|
|
@ -981,7 +1038,11 @@ pub fn update_instance_settings(conn: &Connection, settings: &InstanceSettings)
|
|||
ublock_enabled = ?5,
|
||||
cookie_ext_enabled = ?6,
|
||||
modal_closer_enabled = ?7,
|
||||
reorder_children_role_bits = ?8
|
||||
reorder_children_role_bits = ?8,
|
||||
title_model_anthropic_http = ?9,
|
||||
title_model_openai_compatible = ?10,
|
||||
title_model_claude_cli = ?11,
|
||||
title_model_codex_cli = ?12
|
||||
WHERE id = 1",
|
||||
params![
|
||||
settings.public_index_enabled as i64,
|
||||
|
|
@ -992,6 +1053,10 @@ pub fn update_instance_settings(conn: &Connection, settings: &InstanceSettings)
|
|||
settings.cookie_ext_enabled as i64,
|
||||
settings.modal_closer_enabled as i64,
|
||||
settings.reorder_children_role_bits as i64,
|
||||
settings.title_model_anthropic_http,
|
||||
settings.title_model_openai_compatible,
|
||||
settings.title_model_claude_cli,
|
||||
settings.title_model_codex_cli,
|
||||
],
|
||||
)?;
|
||||
Ok(())
|
||||
|
|
@ -1121,6 +1186,46 @@ pub fn update_entry_title(conn: &Connection, entry_uid: &str, title: Option<&str
|
|||
Ok(n > 0)
|
||||
}
|
||||
|
||||
/// An `x`/`tweet` entry whose title may be a legacy bare-link auto title.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct TweetTitleCandidate {
|
||||
pub id: i64,
|
||||
pub title: String,
|
||||
pub source_metadata_json: String,
|
||||
}
|
||||
|
||||
/// Root or child `x`/`tweet` entries whose title starts with "http" (bare-link auto titles).
|
||||
pub fn list_bare_link_tweet_titles(conn: &Connection) -> Result<Vec<TweetTitleCandidate>> {
|
||||
let mut stmt = conn.prepare(
|
||||
"SELECT id, title, source_metadata_json FROM archived_entries
|
||||
WHERE source_kind = 'x' AND entity_kind = 'tweet' AND title LIKE 'http%'
|
||||
ORDER BY id",
|
||||
)?;
|
||||
let rows = stmt.query_map([], |row| {
|
||||
Ok(TweetTitleCandidate {
|
||||
id: row.get(0)?,
|
||||
title: row.get(1)?,
|
||||
source_metadata_json: row.get(2)?,
|
||||
})
|
||||
})?;
|
||||
Ok(rows.collect::<rusqlite::Result<Vec<_>>>()?)
|
||||
}
|
||||
|
||||
/// Compare-and-set title update: only writes when the stored title still equals
|
||||
/// `expected`, so a concurrent rename always wins. `Ok(true)` iff one row changed.
|
||||
pub fn replace_entry_title_if_unchanged(
|
||||
conn: &Connection,
|
||||
entry_id: i64,
|
||||
expected: &str,
|
||||
new_title: &str,
|
||||
) -> Result<bool> {
|
||||
let n = conn.execute(
|
||||
"UPDATE archived_entries SET title = ?1 WHERE id = ?2 AND title = ?3",
|
||||
params![new_title, entry_id, expected],
|
||||
)?;
|
||||
Ok(n == 1)
|
||||
}
|
||||
|
||||
/// Outcome of [`reorder_child_entries`]; the server maps it to 204/404/400.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum ReorderChildrenOutcome {
|
||||
|
|
@ -1667,6 +1772,36 @@ pub fn entry_id_for_uid(conn: &Connection, entry_uid: &str) -> Result<Option<i64
|
|||
.map_err(Into::into)
|
||||
}
|
||||
|
||||
/// An entry's id plus the source identity fields needed to re-fetch it.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct EntrySourceInfo {
|
||||
pub entry_id: i64,
|
||||
pub source_kind: String,
|
||||
pub entity_kind: String,
|
||||
pub canonical_url: Option<String>,
|
||||
}
|
||||
|
||||
/// Looks up `entry_uid` with its source identity's canonical URL. `Ok(None)` if absent.
|
||||
pub fn entry_source_info(conn: &Connection, entry_uid: &str) -> Result<Option<EntrySourceInfo>> {
|
||||
conn.query_row(
|
||||
"SELECT e.id, e.source_kind, e.entity_kind, si.canonical_url
|
||||
FROM archived_entries e
|
||||
JOIN source_identities si ON si.id = e.source_identity_id
|
||||
WHERE e.entry_uid = ?1",
|
||||
[entry_uid],
|
||||
|row| {
|
||||
Ok(EntrySourceInfo {
|
||||
entry_id: row.get(0)?,
|
||||
source_kind: row.get(1)?,
|
||||
entity_kind: row.get(2)?,
|
||||
canonical_url: row.get(3)?,
|
||||
})
|
||||
},
|
||||
)
|
||||
.optional()
|
||||
.map_err(Into::into)
|
||||
}
|
||||
|
||||
/// Creates a fresh pending summary attempt for one cache key.
|
||||
///
|
||||
/// Attempts are intentionally not unique by cache key: a forced regeneration
|
||||
|
|
@ -1751,6 +1886,20 @@ pub fn update_entry_summary_status(
|
|||
Ok(())
|
||||
}
|
||||
|
||||
/// Replaces a summary row's input digest (used once a deferred input is built).
|
||||
/// Also bumps `updated_at`.
|
||||
pub fn update_entry_summary_input_sha256(
|
||||
conn: &Connection,
|
||||
summary_uid: &str,
|
||||
input_sha256: &str,
|
||||
) -> Result<()> {
|
||||
conn.execute(
|
||||
"UPDATE entry_summaries SET input_sha256 = ?1, updated_at = ?2 WHERE summary_uid = ?3",
|
||||
params![input_sha256, now_timestamp(), summary_uid],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Returns one summary by its public uid.
|
||||
pub fn get_entry_summary_by_uid(
|
||||
conn: &Connection,
|
||||
|
|
@ -2284,6 +2433,19 @@ pub fn has_active_capture_jobs(conn: &Connection) -> Result<bool> {
|
|||
Ok(n > 0)
|
||||
}
|
||||
|
||||
/// Returns `true` while a summary-time subtitle fetch is in flight (a
|
||||
/// `pending`/`running` summary row still carrying the placeholder digest).
|
||||
/// Like a capture, the fetch moves files into `raw/` before writing DB rows.
|
||||
pub fn has_pending_subtitle_fetches(conn: &Connection) -> Result<bool> {
|
||||
let n: i64 = conn.query_row(
|
||||
"SELECT COUNT(*) FROM entry_summaries
|
||||
WHERE status IN ('pending', 'running') AND input_sha256 = ?1",
|
||||
[crate::summarizer::SUBTITLE_FETCH_PENDING_INPUT_SHA256],
|
||||
|row| row.get(0),
|
||||
)?;
|
||||
Ok(n > 0)
|
||||
}
|
||||
|
||||
/// Returns `(id, raw_relpath, byte_size)` for every blob row not referenced by any
|
||||
/// `entry_artifacts.blob_id`. These DB rows are safe to delete regardless of whether
|
||||
/// a disk file still exists at their `raw_relpath`.
|
||||
|
|
@ -2436,6 +2598,60 @@ pub fn add_entry_artifact(conn: &Connection, artifact: &NewArtifact) -> Result<i
|
|||
Ok(conn.last_insert_rowid())
|
||||
}
|
||||
|
||||
/// One artifact of a given role, with its blob MIME type when it has a blob.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct RoleArtifact {
|
||||
pub id: i64,
|
||||
pub relpath: String,
|
||||
pub mime_type: Option<String>,
|
||||
pub metadata_json: Option<String>,
|
||||
}
|
||||
|
||||
/// Lists an entry's artifacts with `role`, in insertion (id) order.
|
||||
pub fn list_entry_artifacts_by_role(
|
||||
conn: &Connection,
|
||||
entry_id: i64,
|
||||
role: &str,
|
||||
) -> Result<Vec<RoleArtifact>> {
|
||||
let mut stmt = conn.prepare(
|
||||
"SELECT ea.id, ea.relpath, b.mime_type, ea.metadata_json
|
||||
FROM entry_artifacts ea
|
||||
LEFT JOIN blobs b ON b.id = ea.blob_id
|
||||
WHERE ea.entry_id = ?1 AND ea.artifact_role = ?2
|
||||
ORDER BY ea.id ASC",
|
||||
)?;
|
||||
let rows = stmt
|
||||
.query_map(params![entry_id, role], |row| {
|
||||
Ok(RoleArtifact {
|
||||
id: row.get(0)?,
|
||||
relpath: row.get(1)?,
|
||||
mime_type: row.get(2)?,
|
||||
metadata_json: row.get(3)?,
|
||||
})
|
||||
})?
|
||||
.collect::<rusqlite::Result<Vec<_>>>()?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
/// True if the entry already has an artifact with `role` pointing at `blob_id`.
|
||||
/// `entry_artifacts` has no uniqueness constraint, so callers dedupe with this.
|
||||
pub fn entry_has_artifact_blob(
|
||||
conn: &Connection,
|
||||
entry_id: i64,
|
||||
role: &str,
|
||||
blob_id: i64,
|
||||
) -> Result<bool> {
|
||||
let exists: bool = conn.query_row(
|
||||
"SELECT EXISTS(
|
||||
SELECT 1 FROM entry_artifacts
|
||||
WHERE entry_id = ?1 AND artifact_role = ?2 AND blob_id = ?3
|
||||
)",
|
||||
params![entry_id, role, blob_id],
|
||||
|row| row.get(0),
|
||||
)?;
|
||||
Ok(exists)
|
||||
}
|
||||
|
||||
pub fn remove_entry_tag_assignment(conn: &Connection, entry_id: i64, tag_id: i64) -> Result<()> {
|
||||
conn.execute(
|
||||
"DELETE FROM entry_tag_assignments WHERE entry_id = ?1 AND tag_id = ?2",
|
||||
|
|
@ -3301,6 +3517,26 @@ mod tests {
|
|||
.unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn replace_entry_title_if_unchanged_is_compare_and_set() {
|
||||
let conn = conn();
|
||||
let entry = create_entry_fixture(&conn, "private", None, None);
|
||||
update_entry_title(&conn, &entry.entry_uid, Some("a")).unwrap();
|
||||
let title = |conn: &Connection| -> String {
|
||||
conn.query_row(
|
||||
"SELECT title FROM archived_entries WHERE id = ?1",
|
||||
[entry.id],
|
||||
|row| row.get(0),
|
||||
)
|
||||
.unwrap()
|
||||
};
|
||||
|
||||
assert!(!replace_entry_title_if_unchanged(&conn, entry.id, "b", "c").unwrap());
|
||||
assert_eq!(title(&conn), "a");
|
||||
assert!(replace_entry_title_if_unchanged(&conn, entry.id, "a", "c").unwrap());
|
||||
assert_eq!(title(&conn), "c");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn schema_defaults_public_settings_to_private() {
|
||||
let conn = conn();
|
||||
|
|
@ -3990,6 +4226,29 @@ mod tests {
|
|||
assert!(s.modal_closer_enabled);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn instance_settings_title_models_migrate_and_round_trip() {
|
||||
let conn = Connection::open_in_memory().unwrap();
|
||||
conn.execute_batch(
|
||||
"CREATE TABLE instance_settings (id INTEGER PRIMARY KEY CHECK (id = 1), public_index_enabled INTEGER NOT NULL DEFAULT 0, public_entry_content_enabled INTEGER NOT NULL DEFAULT 0, public_archive_submission_enabled INTEGER NOT NULL DEFAULT 0, default_entry_visibility INTEGER NOT NULL DEFAULT 2);
|
||||
INSERT INTO instance_settings (id) VALUES (1);",
|
||||
)
|
||||
.unwrap();
|
||||
initialize_auth_schema(&conn).unwrap();
|
||||
initialize_auth_schema(&conn).unwrap();
|
||||
let mut s = get_instance_settings(&conn).unwrap();
|
||||
assert_eq!(s.title_model_claude_cli, None);
|
||||
assert_eq!(s.title_model_override("claude_cli"), None);
|
||||
*s.title_model_slot_mut("claude_cli").unwrap() = Some("sonnet".into());
|
||||
s.title_model_codex_cli = Some(" ".into());
|
||||
update_instance_settings(&conn, &s).unwrap();
|
||||
let s = get_instance_settings(&conn).unwrap();
|
||||
assert_eq!(s.title_model_override("claude_cli"), Some("sonnet"));
|
||||
assert_eq!(s.title_model_override("codex_cli"), None);
|
||||
assert_eq!(s.title_model_override("gemini"), None);
|
||||
assert_eq!(s.title_model_anthropic_http, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn can_reorder_children_intersects_mask() {
|
||||
let conn = make_auth_conn_for_mgmt();
|
||||
|
|
@ -4928,6 +5187,22 @@ mod tests {
|
|||
assert!(rec.completed_at.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn has_pending_subtitle_fetches_tracks_placeholder_rows() {
|
||||
let c = conn();
|
||||
let entry = create_entry_fixture(&c, "private", None, None);
|
||||
let placeholder = crate::summarizer::SUBTITLE_FETCH_PENDING_INPUT_SHA256;
|
||||
upsert_pending_entry_summary(&c, entry.id, "codex_cli", None, "v1", "real").unwrap();
|
||||
assert!(!has_pending_subtitle_fetches(&c).unwrap());
|
||||
let uid =
|
||||
upsert_pending_entry_summary(&c, entry.id, "codex_cli", None, "v1", placeholder).unwrap();
|
||||
assert!(has_pending_subtitle_fetches(&c).unwrap());
|
||||
update_entry_summary_status(&c, &uid, "running", None, None).unwrap();
|
||||
assert!(has_pending_subtitle_fetches(&c).unwrap());
|
||||
update_entry_summary_status(&c, &uid, "failed", None, Some("x")).unwrap();
|
||||
assert!(!has_pending_subtitle_fetches(&c).unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn regenerating_a_completed_summary_creates_a_new_attempt_and_preserves_completion() {
|
||||
let c = conn();
|
||||
|
|
@ -5275,4 +5550,119 @@ mod tests {
|
|||
assert_eq!(position_of(&c, b.id), Some(1));
|
||||
assert_eq!(position_of(&c, c3.id), Some(2));
|
||||
}
|
||||
|
||||
fn test_blob(conn: &Connection, sha: &str, mime: &str) -> i64 {
|
||||
upsert_blob(
|
||||
conn,
|
||||
&BlobRecord {
|
||||
sha256: sha.to_string(),
|
||||
byte_size: 10,
|
||||
mime_type: Some(mime.to_string()),
|
||||
extension: None,
|
||||
raw_relpath: format!("raw/{sha}"),
|
||||
},
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn test_artifact(conn: &Connection, entry_id: i64, role: &str, blob_id: Option<i64>, relpath: &str) -> i64 {
|
||||
add_entry_artifact(
|
||||
conn,
|
||||
&NewArtifact {
|
||||
entry_id,
|
||||
artifact_role: role.to_string(),
|
||||
storage_area: "raw".to_string(),
|
||||
relpath: relpath.to_string(),
|
||||
blob_id,
|
||||
logical_path: None,
|
||||
metadata_json: Some(format!("{{\"r\":\"{relpath}\"}}")),
|
||||
},
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entry_source_info_joins_canonical_url() {
|
||||
let conn = conn();
|
||||
let entry = create_entry_fixture(&conn, "private", None, None);
|
||||
let info = entry_source_info(&conn, &entry.entry_uid).unwrap().unwrap();
|
||||
assert_eq!(
|
||||
info,
|
||||
EntrySourceInfo {
|
||||
entry_id: entry.id,
|
||||
source_kind: "youtube".to_string(),
|
||||
entity_kind: "video".to_string(),
|
||||
canonical_url: Some("https://youtube.com/watch?v=video-1".to_string()),
|
||||
}
|
||||
);
|
||||
assert!(entry_source_info(&conn, "entry_missing").unwrap().is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_entry_artifacts_by_role_orders_by_id() {
|
||||
let conn = conn();
|
||||
let entry = create_entry_fixture(&conn, "private", None, None);
|
||||
let vtt = test_blob(&conn, "aa11", "text/vtt");
|
||||
let first = test_artifact(&conn, entry.id, "subtitle", Some(vtt), "raw/a/a/aa11.vtt");
|
||||
let _media = test_artifact(&conn, entry.id, "primary_media", None, "raw/m.mp4");
|
||||
let second = test_artifact(&conn, entry.id, "subtitle", None, "raw/b.srt");
|
||||
|
||||
let rows = list_entry_artifacts_by_role(&conn, entry.id, "subtitle").unwrap();
|
||||
assert_eq!(
|
||||
rows,
|
||||
vec![
|
||||
RoleArtifact {
|
||||
id: first,
|
||||
relpath: "raw/a/a/aa11.vtt".to_string(),
|
||||
mime_type: Some("text/vtt".to_string()),
|
||||
metadata_json: Some("{\"r\":\"raw/a/a/aa11.vtt\"}".to_string()),
|
||||
},
|
||||
RoleArtifact {
|
||||
id: second,
|
||||
relpath: "raw/b.srt".to_string(),
|
||||
mime_type: None,
|
||||
metadata_json: Some("{\"r\":\"raw/b.srt\"}".to_string()),
|
||||
},
|
||||
]
|
||||
);
|
||||
assert!(list_entry_artifacts_by_role(&conn, entry.id, "favicon").unwrap().is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entry_has_artifact_blob_matches_role_and_blob() {
|
||||
let conn = conn();
|
||||
let entry = create_entry_fixture(&conn, "private", None, None);
|
||||
let other = create_entry_fixture(&conn, "private", None, None);
|
||||
let blob = test_blob(&conn, "bb22", "text/vtt");
|
||||
let unrelated = test_blob(&conn, "cc33", "text/vtt");
|
||||
test_artifact(&conn, entry.id, "subtitle", Some(blob), "raw/bb22.vtt");
|
||||
|
||||
assert!(entry_has_artifact_blob(&conn, entry.id, "subtitle", blob).unwrap());
|
||||
assert!(!entry_has_artifact_blob(&conn, entry.id, "primary_media", blob).unwrap());
|
||||
assert!(!entry_has_artifact_blob(&conn, entry.id, "subtitle", unrelated).unwrap());
|
||||
assert!(!entry_has_artifact_blob(&conn, other.id, "subtitle", blob).unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn update_entry_summary_input_sha256_updates_row() {
|
||||
let conn = conn();
|
||||
let entry = create_entry_fixture(&conn, "private", None, None);
|
||||
let uid = upsert_pending_entry_summary(
|
||||
&conn,
|
||||
entry.id,
|
||||
"codex_cli",
|
||||
None,
|
||||
"v1",
|
||||
"pending-subtitle-fetch",
|
||||
)
|
||||
.unwrap();
|
||||
let before = get_entry_summary_by_uid(&conn, &uid).unwrap().unwrap();
|
||||
|
||||
let digest = "ab".repeat(32);
|
||||
update_entry_summary_input_sha256(&conn, &uid, &digest).unwrap();
|
||||
let after = get_entry_summary_by_uid(&conn, &uid).unwrap().unwrap();
|
||||
assert_eq!(after.input_sha256, digest);
|
||||
assert_eq!(after.status, "pending");
|
||||
assert!(after.updated_at >= before.updated_at);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
328
crates/archivr-core/src/downloader/deno_install.rs
Normal file
328
crates/archivr-core/src/downloader/deno_install.rs
Normal file
|
|
@ -0,0 +1,328 @@
|
|||
//! Deno half of the shared yt-dlp tools update (CLI `archivr yt-dlp update` and the admin
|
||||
//! UI): installs the latest official Deno release into
|
||||
//! `<state_dir>/deno/deno`, the JS runtime yt-dlp uses to solve YouTube's challenges.
|
||||
//!
|
||||
//! The flow mirrors the yt-dlp zipapp install: download, extract into a staging file,
|
||||
//! verify it runs (`--version` must report exactly the release version), then rename it
|
||||
//! over the target so a concurrently-running archivr never sees a half-written binary.
|
||||
//! The release `.sha256sum` is not checked: it comes from the same TLS origin as the zip,
|
||||
//! and the zip's CRC32 already catches corruption.
|
||||
|
||||
use anyhow::{bail, Context, Result};
|
||||
use super::js_runtime::{
|
||||
parse_deno_version_output, pinned_deno, probe_deno_version, state_dir_deno, DenoVersion,
|
||||
MIN_DENO_VERSION,
|
||||
};
|
||||
use std::{
|
||||
env, fs,
|
||||
io::{self, Cursor},
|
||||
path::Path,
|
||||
process::Command,
|
||||
time::Duration,
|
||||
};
|
||||
|
||||
/// GitHub release metadata endpoint for the upstream Deno project.
|
||||
pub const DENO_LATEST_RELEASE: &str =
|
||||
"https://api.github.com/repos/denoland/deno/releases/latest";
|
||||
|
||||
/// The Deno zip is ~40 MB; reqwest's blocking client defaults to a 30s total timeout,
|
||||
/// which is too short on slow links. Applies to the zip download only.
|
||||
const DENO_DOWNLOAD_TIMEOUT: Duration = Duration::from_secs(600);
|
||||
|
||||
/// Why a staged prebuilt Deno can fail to spawn even though the file exists: the official
|
||||
/// binaries are dynamically linked against a glibc loader NixOS doesn't provide.
|
||||
const NO_LOADER: &str = "prebuilt deno cannot execute on this host \
|
||||
(missing dynamic loader — on NixOS enable programs.nix-ld)";
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct DenoRelease {
|
||||
pub tag: String,
|
||||
pub version: DenoVersion,
|
||||
pub download_url: String,
|
||||
}
|
||||
|
||||
/// Official release asset for a `std::env::consts::{OS, ARCH}` pair.
|
||||
pub fn deno_release_asset(os: &str, arch: &str) -> Option<&'static str> {
|
||||
match (os, arch) {
|
||||
("macos", "aarch64") => Some("deno-aarch64-apple-darwin.zip"),
|
||||
("macos", "x86_64") => Some("deno-x86_64-apple-darwin.zip"),
|
||||
("linux", "x86_64") => Some("deno-x86_64-unknown-linux-gnu.zip"),
|
||||
("linux", "aarch64") => Some("deno-aarch64-unknown-linux-gnu.zip"),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Extracts the version and download URL from a GitHub "latest release" response.
|
||||
/// The URL is built from the tag and asset name rather than taken from the response.
|
||||
pub fn parse_deno_release(json: &serde_json::Value, asset: &str) -> Result<DenoRelease> {
|
||||
let tag = json
|
||||
.get("tag_name")
|
||||
.and_then(serde_json::Value::as_str)
|
||||
.context("GitHub releases API response had no tag_name")?;
|
||||
// The tag ends up in a URL path; only accept plain version-ish characters.
|
||||
if tag.is_empty()
|
||||
|| !tag
|
||||
.chars()
|
||||
.all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '-' | '+'))
|
||||
{
|
||||
bail!("unexpected deno release tag {tag:?}");
|
||||
}
|
||||
let version = DenoVersion::parse(tag)
|
||||
.with_context(|| format!("could not parse a version from deno release tag {tag:?}"))?;
|
||||
let has_asset = json
|
||||
.get("assets")
|
||||
.and_then(serde_json::Value::as_array)
|
||||
.is_some_and(|assets| {
|
||||
assets
|
||||
.iter()
|
||||
.any(|a| a.get("name").and_then(serde_json::Value::as_str) == Some(asset))
|
||||
});
|
||||
if !has_asset {
|
||||
bail!("deno release {tag} has no {asset} asset");
|
||||
}
|
||||
Ok(DenoRelease {
|
||||
tag: tag.to_string(),
|
||||
version,
|
||||
download_url: format!(
|
||||
"https://github.com/denoland/deno/releases/download/{tag}/{asset}"
|
||||
),
|
||||
})
|
||||
}
|
||||
|
||||
/// Installs or updates Deno in the state dir. `Ok` carries a one-line human outcome.
|
||||
pub fn install_deno(client: &reqwest::blocking::Client, log: &mut dyn FnMut(&str)) -> Result<String> {
|
||||
let target = state_dir_deno().context("could not determine a state directory (is $HOME set?)")?;
|
||||
let dir = target
|
||||
.parent()
|
||||
.context("state-dir deno path has no parent directory")?;
|
||||
let staging = dir.join("deno.new");
|
||||
|
||||
let (os, arch) = (env::consts::OS, env::consts::ARCH);
|
||||
let asset = deno_release_asset(os, arch)
|
||||
.with_context(|| format!("unsupported platform {os}/{arch}"))?;
|
||||
|
||||
let body = client
|
||||
.get(DENO_LATEST_RELEASE)
|
||||
.send()
|
||||
.context("failed to reach the GitHub releases API")?
|
||||
.error_for_status()
|
||||
.context("GitHub releases API returned an error")?
|
||||
.text()
|
||||
.context("failed to read the GitHub releases API response")?;
|
||||
let json: serde_json::Value =
|
||||
serde_json::from_str(&body).context("GitHub releases API returned invalid JSON")?;
|
||||
let release = parse_deno_release(&json, asset)?;
|
||||
|
||||
if let Some(installed) = probe_deno_version(&target).filter(|v| *v >= release.version) {
|
||||
return Ok(format!("deno {installed} already installed at {}", target.display()));
|
||||
}
|
||||
|
||||
log(&format!("Downloading deno {}…", release.version));
|
||||
let bytes = client
|
||||
.get(&release.download_url)
|
||||
.timeout(DENO_DOWNLOAD_TIMEOUT)
|
||||
.send()
|
||||
.with_context(|| format!("failed to download {}", release.download_url))?
|
||||
.error_for_status()
|
||||
.with_context(|| format!("download of {} failed", release.download_url))?
|
||||
.bytes()
|
||||
.context("failed to read the downloaded deno zip")?;
|
||||
|
||||
fs::create_dir_all(dir).with_context(|| format!("failed to create {}", dir.display()))?;
|
||||
|
||||
let staged = extract_deno(&bytes, &staging)
|
||||
.and_then(|()| verify_staged(&staging, release.version))
|
||||
.and_then(|runs| {
|
||||
if runs {
|
||||
fs::rename(&staging, &target)
|
||||
.with_context(|| format!("failed to install {}", target.display()))?;
|
||||
}
|
||||
Ok(runs)
|
||||
});
|
||||
let runs = staged.inspect_err(|_| {
|
||||
let _ = fs::remove_file(&staging);
|
||||
})?;
|
||||
|
||||
if !runs {
|
||||
let _ = fs::remove_file(&staging);
|
||||
let pinned_usable = pinned_deno()
|
||||
.and_then(|p| probe_deno_version(&p))
|
||||
.is_some_and(|v| v >= MIN_DENO_VERSION);
|
||||
if pinned_usable {
|
||||
return Ok(format!("skipped: {NO_LOADER}; using pinned ARCHIVR_DENO"));
|
||||
}
|
||||
bail!("{NO_LOADER}, and no usable pinned ARCHIVR_DENO is set");
|
||||
}
|
||||
|
||||
Ok(format!("installed deno {} to {}", release.version, target.display()))
|
||||
}
|
||||
|
||||
/// Writes the zip's `deno` entry to `staging` and makes it executable.
|
||||
fn extract_deno(zip_bytes: &[u8], staging: &Path) -> Result<()> {
|
||||
let mut archive =
|
||||
zip::ZipArchive::new(Cursor::new(zip_bytes)).context("downloaded deno zip is invalid")?;
|
||||
let mut entry = archive
|
||||
.by_name("deno")
|
||||
.context("downloaded deno zip has no `deno` entry")?;
|
||||
let mut out = fs::File::create(staging)
|
||||
.with_context(|| format!("failed to create {}", staging.display()))?;
|
||||
io::copy(&mut entry, &mut out)
|
||||
.with_context(|| format!("failed to extract deno to {}", staging.display()))?;
|
||||
out.sync_all()
|
||||
.with_context(|| format!("failed to flush {}", staging.display()))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
fs::set_permissions(staging, fs::Permissions::from_mode(0o755))
|
||||
.with_context(|| format!("failed to chmod +x {}", staging.display()))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Runs `staging --version` and requires it to report exactly `expected`.
|
||||
///
|
||||
/// `Ok(false)` means the binary exists but the OS could not execute it at all
|
||||
/// (spawn failed with `NotFound` — e.g. the ELF interpreter is missing on NixOS);
|
||||
/// any other failure is an error.
|
||||
fn verify_staged(staging: &Path, expected: DenoVersion) -> Result<bool> {
|
||||
let output = match Command::new(staging).arg("--version").output() {
|
||||
Ok(output) => output,
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound && staging.is_file() => return Ok(false),
|
||||
Err(e) => {
|
||||
return Err(e).with_context(|| format!("failed to run {} --version", staging.display()));
|
||||
}
|
||||
};
|
||||
let got = output
|
||||
.status
|
||||
.success()
|
||||
.then(|| parse_deno_version_output(&String::from_utf8_lossy(&output.stdout)))
|
||||
.flatten();
|
||||
if got != Some(expected) {
|
||||
let got = got.map_or_else(|| "no parseable version".to_string(), |v| v.to_string());
|
||||
bail!("downloaded deno failed verification (expected {expected}, got {got})");
|
||||
}
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{deno_release_asset, parse_deno_release, verify_staged};
|
||||
#[cfg(unix)]
|
||||
use crate::downloader::write_script;
|
||||
use crate::downloader::js_runtime::DenoVersion;
|
||||
use serde_json::json;
|
||||
|
||||
#[test]
|
||||
fn supported_platforms_map_to_assets() {
|
||||
assert_eq!(
|
||||
deno_release_asset("macos", "aarch64"),
|
||||
Some("deno-aarch64-apple-darwin.zip")
|
||||
);
|
||||
assert_eq!(
|
||||
deno_release_asset("macos", "x86_64"),
|
||||
Some("deno-x86_64-apple-darwin.zip")
|
||||
);
|
||||
assert_eq!(
|
||||
deno_release_asset("linux", "x86_64"),
|
||||
Some("deno-x86_64-unknown-linux-gnu.zip")
|
||||
);
|
||||
assert_eq!(
|
||||
deno_release_asset("linux", "aarch64"),
|
||||
Some("deno-aarch64-unknown-linux-gnu.zip")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unsupported_platforms_have_no_asset() {
|
||||
assert_eq!(deno_release_asset("windows", "x86_64"), None);
|
||||
assert_eq!(deno_release_asset("linux", "riscv64"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn release_json_yields_version_and_url() {
|
||||
let json = json!({
|
||||
"tag_name": "v2.9.7",
|
||||
"assets": [
|
||||
{"name": "deno-x86_64-unknown-linux-gnu.zip"},
|
||||
{"name": "deno-aarch64-apple-darwin.zip"},
|
||||
],
|
||||
});
|
||||
let release = parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").unwrap();
|
||||
assert_eq!(release.tag, "v2.9.7");
|
||||
assert_eq!(
|
||||
release.version,
|
||||
DenoVersion { major: 2, minor: 9, patch: 7 }
|
||||
);
|
||||
assert_eq!(
|
||||
release.download_url,
|
||||
"https://github.com/denoland/deno/releases/download/v2.9.7/deno-aarch64-apple-darwin.zip"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_asset_is_an_error_naming_it() {
|
||||
let json = json!({
|
||||
"tag_name": "v2.9.7",
|
||||
"assets": [{"name": "deno-x86_64-unknown-linux-gnu.zip"}],
|
||||
});
|
||||
let err = parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").unwrap_err();
|
||||
assert!(
|
||||
err.to_string().contains("deno-aarch64-apple-darwin.zip"),
|
||||
"{err:#}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_or_bad_tag_is_an_error() {
|
||||
let assets = json!([{"name": "deno-aarch64-apple-darwin.zip"}]);
|
||||
for json in [
|
||||
json!({"assets": assets}),
|
||||
json!({"tag_name": 297, "assets": assets}),
|
||||
json!({"tag_name": "nightly", "assets": assets}),
|
||||
json!({"tag_name": "v2.9.7/../../evil", "assets": assets}),
|
||||
] {
|
||||
assert!(
|
||||
parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").is_err(),
|
||||
"{json}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn staged_binary_must_report_the_release_version() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let expected = DenoVersion { major: 2, minor: 9, patch: 7 };
|
||||
|
||||
// A fresh path per script: rewriting one path that was just exec'd invites ETXTBSY.
|
||||
let good = tmp.path().join("good/deno.new");
|
||||
write_script(&good, "#!/bin/sh\necho 'deno 2.9.7 (stable, release, test)'\n");
|
||||
assert!(verify_staged(&good, expected).unwrap());
|
||||
|
||||
let wrong = tmp.path().join("wrong/deno.new");
|
||||
write_script(&wrong, "#!/bin/sh\necho 'deno 2.9.6 (stable, release, test)'\n");
|
||||
let err = verify_staged(&wrong, expected).unwrap_err();
|
||||
assert!(err.to_string().contains("expected 2.9.7, got 2.9.6"), "{err:#}");
|
||||
|
||||
let failing = tmp.path().join("failing/deno.new");
|
||||
write_script(&failing, "#!/bin/sh\nexit 1\n");
|
||||
assert!(verify_staged(&failing, expected).is_err());
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn existing_but_unexecutable_binary_is_reported_as_cannot_run() {
|
||||
// A script whose interpreter is missing fails to spawn with NotFound even though
|
||||
// the file exists — the same shape as a glibc ELF on NixOS without nix-ld.
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let staged = tmp.path().join("deno.new");
|
||||
write_script(&staged, "#!/nonexistent/ld-linux.so\n");
|
||||
let expected = DenoVersion { major: 2, minor: 9, patch: 7 };
|
||||
assert!(!verify_staged(&staged, expected).unwrap());
|
||||
|
||||
// A genuinely missing file is an error, not "cannot execute".
|
||||
let missing = tmp.path().join("missing");
|
||||
assert!(verify_staged(&missing, expected).is_err());
|
||||
}
|
||||
}
|
||||
684
crates/archivr-core/src/downloader/js_runtime.rs
Normal file
684
crates/archivr-core/src/downloader/js_runtime.rs
Normal file
|
|
@ -0,0 +1,684 @@
|
|||
//! JavaScript runtime resolution for yt-dlp.
|
||||
//!
|
||||
//! yt-dlp needs a JS runtime to solve YouTube's challenges (EJS). Without one, YouTube
|
||||
//! downloads may fail with HTTP 403. This module picks the runtime archivr passes to every
|
||||
//! yt-dlp process via `--js-runtimes`:
|
||||
//!
|
||||
//! 1. `ARCHIVR_JS_RUNTIME` (forced, `RUNTIME[:ABS_PATH]`; skips resolution and version checks).
|
||||
//! 2. The newest Deno >= [`MIN_DENO_VERSION`] among the pinned `ARCHIVR_DENO` and the
|
||||
//! state-dir copy (`<state_dir>/deno/deno`); an exact tie goes to the state dir.
|
||||
//! 3. `deno` on `PATH`, if new enough.
|
||||
//! 4. Nothing (a warning is printed once per resolution: first use and each refresh).
|
||||
//!
|
||||
//! Only Deno is ever chosen automatically; Node, Bun and QuickJS are used only when forced.
|
||||
|
||||
use std::{
|
||||
env,
|
||||
ffi::{OsStr, OsString},
|
||||
fmt,
|
||||
path::{Path, PathBuf},
|
||||
process::Command,
|
||||
sync::RwLock,
|
||||
};
|
||||
|
||||
use super::ytdlp::state_dir;
|
||||
|
||||
/// Forced runtime override, `RUNTIME[:ABS_PATH]` with RUNTIME one of deno|node|bun|quickjs.
|
||||
pub const JS_RUNTIME_ENV: &str = "ARCHIVR_JS_RUNTIME";
|
||||
/// Pinned Deno binary (set by the Nix wrappers and the Docker image).
|
||||
pub const DENO_ENV: &str = "ARCHIVR_DENO";
|
||||
/// Oldest Deno yt-dlp's EJS solver supports.
|
||||
pub const MIN_DENO_VERSION: DenoVersion = DenoVersion { major: 2, minor: 3, patch: 0 };
|
||||
|
||||
/// Cached choice: outer `None` = not resolved yet. Swapped by [`refresh_js_runtime`].
|
||||
static RESOLVED_JS_RUNTIME: RwLock<Option<Option<JsRuntime>>> = RwLock::new(None);
|
||||
|
||||
/// Resolves (uncached) and prints the warnings; runs on first use and on each refresh.
|
||||
fn resolve_js_runtime_logged() -> Option<JsRuntime> {
|
||||
if let Err(reason) = forced_js_runtime() {
|
||||
let raw = env::var_os(JS_RUNTIME_ENV).unwrap_or_default();
|
||||
eprintln!("warn: ignoring {JS_RUNTIME_ENV}={raw:?}: {reason}");
|
||||
}
|
||||
let resolved = resolve_js_runtime_with_role().map(|(_, rt)| rt);
|
||||
if resolved.is_none() {
|
||||
eprintln!(
|
||||
"warn: no JavaScript runtime for yt-dlp (need deno >= {MIN_DENO_VERSION} via \
|
||||
{DENO_ENV}, the state dir, or PATH; or set {JS_RUNTIME_ENV}) — YouTube \
|
||||
downloads may fail with HTTP 403; run `archivr yt-dlp update` to install deno"
|
||||
);
|
||||
}
|
||||
resolved
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum JsRuntimeKind {
|
||||
Deno,
|
||||
Node,
|
||||
Bun,
|
||||
QuickJs,
|
||||
}
|
||||
|
||||
impl JsRuntimeKind {
|
||||
/// Runtime name as yt-dlp's `--js-runtimes` expects it.
|
||||
pub fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Self::Deno => "deno",
|
||||
Self::Node => "node",
|
||||
Self::Bun => "bun",
|
||||
Self::QuickJs => "quickjs",
|
||||
}
|
||||
}
|
||||
|
||||
/// ASCII case-insensitive match against the exact allowlist.
|
||||
pub fn from_name(name: &str) -> Option<Self> {
|
||||
[Self::Deno, Self::Node, Self::Bun, Self::QuickJs]
|
||||
.into_iter()
|
||||
.find(|k| k.as_str().eq_ignore_ascii_case(name))
|
||||
}
|
||||
|
||||
/// Executable name yt-dlp looks for when the runtime path is a directory
|
||||
/// (mirrors `_determine_runtime_path` in yt-dlp's JS runtime classes).
|
||||
fn executable_name(self) -> &'static str {
|
||||
match self {
|
||||
Self::QuickJs => "qjs",
|
||||
other => other.as_str(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct JsRuntime {
|
||||
pub kind: JsRuntimeKind,
|
||||
pub path: Option<PathBuf>,
|
||||
}
|
||||
|
||||
impl JsRuntime {
|
||||
/// `kind` or `kind:path`, built without lossy conversion.
|
||||
pub fn spec(&self) -> OsString {
|
||||
let mut spec = OsString::from(self.kind.as_str());
|
||||
if let Some(path) = &self.path {
|
||||
spec.push(":");
|
||||
spec.push(path.as_os_str());
|
||||
}
|
||||
spec
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
|
||||
pub struct DenoVersion {
|
||||
pub major: u64,
|
||||
pub minor: u64,
|
||||
pub patch: u64,
|
||||
}
|
||||
|
||||
impl DenoVersion {
|
||||
/// Parses `2.9.7` or `v2.9.7`; anything after the patch digits (`+abc`, `-rc1`) is ignored.
|
||||
pub fn parse(s: &str) -> Option<Self> {
|
||||
let s = s.trim();
|
||||
let s = s.strip_prefix('v').unwrap_or(s);
|
||||
let end = s
|
||||
.find(|c: char| !(c.is_ascii_digit() || c == '.'))
|
||||
.unwrap_or(s.len());
|
||||
let mut parts = s[..end].split('.').map(|p| p.parse::<u64>().ok());
|
||||
let version = Self {
|
||||
major: parts.next()??,
|
||||
minor: parts.next()??,
|
||||
patch: parts.next()??,
|
||||
};
|
||||
parts.next().is_none().then_some(version)
|
||||
}
|
||||
}
|
||||
|
||||
impl fmt::Display for DenoVersion {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
write!(f, "{}.{}.{}", self.major, self.minor, self.patch)
|
||||
}
|
||||
}
|
||||
|
||||
/// Validates a `RUNTIME[:ABS_PATH]` spec. The path may be a file or a directory
|
||||
/// (yt-dlp accepts both) but must be absolute and exist.
|
||||
pub fn parse_js_runtime_spec(raw: &str) -> Result<JsRuntime, String> {
|
||||
let raw = raw.trim();
|
||||
let (name, path) = match raw.split_once(':') {
|
||||
Some((name, path)) => (name, Some(path)),
|
||||
None => (raw, None),
|
||||
};
|
||||
let kind = JsRuntimeKind::from_name(name).ok_or_else(|| {
|
||||
format!("unknown runtime {name} (expected deno, node, bun or quickjs)")
|
||||
})?;
|
||||
let path = match path {
|
||||
None => None,
|
||||
Some("") => return Err("empty path after ':'".to_string()),
|
||||
Some(p) => {
|
||||
let p = PathBuf::from(p);
|
||||
if !p.is_absolute() {
|
||||
return Err("path must be absolute".to_string());
|
||||
}
|
||||
if !p.exists() {
|
||||
return Err("path does not exist".to_string());
|
||||
}
|
||||
Some(p)
|
||||
}
|
||||
};
|
||||
Ok(JsRuntime { kind, path })
|
||||
}
|
||||
|
||||
/// Reads `ARCHIVR_JS_RUNTIME`; `Ok(None)` if unset or empty.
|
||||
pub fn forced_js_runtime() -> Result<Option<JsRuntime>, String> {
|
||||
match env::var(JS_RUNTIME_ENV) {
|
||||
Err(env::VarError::NotPresent) => Ok(None),
|
||||
Err(env::VarError::NotUnicode(_)) => Err("value is not valid UTF-8".to_string()),
|
||||
Ok(raw) if raw.trim().is_empty() => Ok(None),
|
||||
Ok(raw) => parse_js_runtime_spec(&raw).map(Some),
|
||||
}
|
||||
}
|
||||
|
||||
/// `ARCHIVR_DENO`, if it points to an existing file.
|
||||
pub fn pinned_deno() -> Option<PathBuf> {
|
||||
env::var_os(DENO_ENV)
|
||||
.filter(|v| !v.is_empty())
|
||||
.map(PathBuf::from)
|
||||
.filter(|p| p.is_file())
|
||||
}
|
||||
|
||||
/// Deno slot inside the state dir (`<state_dir>/deno/deno`); not existence-filtered.
|
||||
pub fn state_dir_deno() -> Option<PathBuf> {
|
||||
state_dir().map(|d| d.join("deno").join("deno"))
|
||||
}
|
||||
|
||||
/// First `dir/name` that is a file, scanning `path_var` like a shell would.
|
||||
pub fn find_on_path(name: &str, path_var: Option<&OsStr>) -> Option<PathBuf> {
|
||||
env::split_paths(path_var?)
|
||||
.filter(|dir| !dir.as_os_str().is_empty())
|
||||
.map(|dir| dir.join(name))
|
||||
.find(|candidate| candidate.is_file())
|
||||
}
|
||||
|
||||
/// `deno` on the process `PATH`.
|
||||
pub fn path_deno() -> Option<PathBuf> {
|
||||
find_on_path("deno", env::var_os("PATH").as_deref())
|
||||
}
|
||||
|
||||
/// Parses `deno --version` output (`deno 2.9.7 (stable, release, ...)` on the first line).
|
||||
pub fn parse_deno_version_output(stdout: &str) -> Option<DenoVersion> {
|
||||
let first = stdout.lines().next()?.trim();
|
||||
let rest = first.strip_prefix("deno ")?;
|
||||
DenoVersion::parse(rest.split_whitespace().next()?)
|
||||
}
|
||||
|
||||
/// Runs `<binary> --version` and parses it; `None` on spawn failure, non-zero exit or junk output.
|
||||
pub fn probe_deno_version(binary: &Path) -> Option<DenoVersion> {
|
||||
let output = Command::new(binary).arg("--version").output().ok()?;
|
||||
if !output.status.success() {
|
||||
return None;
|
||||
}
|
||||
parse_deno_version_output(&String::from_utf8_lossy(&output.stdout))
|
||||
}
|
||||
|
||||
/// Human version string for a (typically forced) runtime. Deno is probed and parsed; other
|
||||
/// runtimes with a path report the first stdout line of `--version`; pathless non-Deno → `None`.
|
||||
/// A directory path gets the runtime's executable name joined, as yt-dlp does.
|
||||
pub fn probe_js_runtime_version(rt: &JsRuntime) -> Option<String> {
|
||||
let binary = match &rt.path {
|
||||
Some(p) if p.is_dir() => p.join(rt.kind.executable_name()),
|
||||
Some(p) => p.clone(),
|
||||
None if rt.kind == JsRuntimeKind::Deno => PathBuf::from("deno"),
|
||||
None => return None,
|
||||
};
|
||||
if rt.kind == JsRuntimeKind::Deno {
|
||||
return probe_deno_version(&binary).map(|v| v.to_string());
|
||||
}
|
||||
let output = Command::new(&binary).arg("--version").output().ok()?;
|
||||
if !output.status.success() {
|
||||
return None;
|
||||
}
|
||||
String::from_utf8_lossy(&output.stdout)
|
||||
.lines()
|
||||
.next()
|
||||
.map(|l| l.trim().to_string())
|
||||
.filter(|l| !l.is_empty())
|
||||
}
|
||||
|
||||
/// Which candidate slot a resolved runtime came from. Several slots can point at the same
|
||||
/// binary (the Nix wrappers set `ARCHIVR_DENO` and also put that Deno on `PATH`), so callers
|
||||
/// that need to name the winner must use the role, not compare paths.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum JsRuntimeRole {
|
||||
/// `ARCHIVR_JS_RUNTIME`.
|
||||
Forced,
|
||||
/// `ARCHIVR_DENO`.
|
||||
Pinned,
|
||||
/// `<state_dir>/deno/deno`.
|
||||
StateDir,
|
||||
/// `deno` on `PATH`.
|
||||
Path,
|
||||
}
|
||||
|
||||
impl JsRuntimeRole {
|
||||
/// Row label used by `archivr yt-dlp status`.
|
||||
pub fn label(self) -> &'static str {
|
||||
match self {
|
||||
Self::Forced => "force (ARCHIVR_JS_RUNTIME)",
|
||||
Self::Pinned => "env (ARCHIVR_DENO)",
|
||||
Self::StateDir => "state-dir",
|
||||
Self::Path => "path (deno)",
|
||||
}
|
||||
}
|
||||
|
||||
/// Stable machine key (API `role`): "force" | "env" | "state-dir" | "path".
|
||||
pub fn key(self) -> &'static str {
|
||||
match self {
|
||||
Self::Forced => "force",
|
||||
Self::Pinned => "env",
|
||||
Self::StateDir => "state-dir",
|
||||
Self::Path => "path",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Existing Deno candidates, pinned first and state-dir last (so ties go to the state dir).
|
||||
pub fn deno_candidates() -> Vec<(JsRuntimeRole, PathBuf)> {
|
||||
[
|
||||
(JsRuntimeRole::Pinned, pinned_deno()),
|
||||
(JsRuntimeRole::StateDir, state_dir_deno()),
|
||||
]
|
||||
.into_iter()
|
||||
.filter_map(|(role, p)| p.filter(|p| p.is_file()).map(|p| (role, p)))
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub(crate) fn resolve_js_runtime_with_path(
|
||||
path_var: Option<&OsStr>,
|
||||
) -> Option<(JsRuntimeRole, JsRuntime)> {
|
||||
if let Ok(Some(rt)) = forced_js_runtime() {
|
||||
return Some((JsRuntimeRole::Forced, rt));
|
||||
}
|
||||
let usable = |p: &Path| probe_deno_version(p).filter(|v| *v >= MIN_DENO_VERSION);
|
||||
// `max_by` keeps the last maximum, so an exact tie goes to the state dir.
|
||||
let (role, best) = deno_candidates()
|
||||
.into_iter()
|
||||
.filter_map(|(role, p)| usable(&p).map(|v| (v, role, p)))
|
||||
.max_by(|(a, ..), (b, ..)| a.cmp(b))
|
||||
.map(|(_, role, p)| (role, p))
|
||||
.or_else(|| {
|
||||
find_on_path("deno", path_var)
|
||||
.filter(|p| usable(p).is_some())
|
||||
.map(|p| (JsRuntimeRole::Path, p))
|
||||
})?;
|
||||
Some((role, JsRuntime { kind: JsRuntimeKind::Deno, path: Some(best) }))
|
||||
}
|
||||
|
||||
/// Resolves without caching or printing, also reporting which candidate slot won
|
||||
/// (used by `archivr yt-dlp status` to mark exactly one row).
|
||||
pub fn resolve_js_runtime_with_role() -> Option<(JsRuntimeRole, JsRuntime)> {
|
||||
resolve_js_runtime_with_path(env::var_os("PATH").as_deref())
|
||||
}
|
||||
|
||||
/// Cached until [`refresh_js_runtime`]; owned clone so a refresh never invalidates a caller.
|
||||
pub fn resolve_js_runtime() -> Option<JsRuntime> {
|
||||
if let Some(v) = RESOLVED_JS_RUNTIME.read().unwrap_or_else(|e| e.into_inner()).as_ref() {
|
||||
return v.clone();
|
||||
}
|
||||
let mut slot = RESOLVED_JS_RUNTIME.write().unwrap_or_else(|e| e.into_inner());
|
||||
slot.get_or_insert_with(resolve_js_runtime_logged).clone()
|
||||
}
|
||||
|
||||
/// Re-resolves (outside the lock) and swaps the cache; called after a successful Deno
|
||||
/// install. Commands already built keep their old runtime.
|
||||
pub fn refresh_js_runtime() -> Option<JsRuntime> {
|
||||
let fresh = resolve_js_runtime_logged();
|
||||
*RESOLVED_JS_RUNTIME.write().unwrap_or_else(|e| e.into_inner()) = Some(fresh.clone());
|
||||
fresh
|
||||
}
|
||||
|
||||
/// yt-dlp arguments selecting `runtime`.
|
||||
///
|
||||
/// yt-dlp builds its runtime map keyed by name, splitting each `--js-runtimes` value on the
|
||||
/// first `:` (`yt_dlp/__init__.py:784-786`), so a later entry for the same name wins:
|
||||
/// `deno:<path>` replaces the default `deno` entry. Non-Deno runtimes are preceded by
|
||||
/// `--no-js-runtimes` (`options.py:460-479`) so a Deno yt-dlp finds on its own can't take
|
||||
/// priority over the forced choice. Each value is one argv element; no shell is involved.
|
||||
pub fn js_runtime_args(runtime: Option<&JsRuntime>) -> Vec<OsString> {
|
||||
let Some(rt) = runtime else {
|
||||
return Vec::new();
|
||||
};
|
||||
let mut args = Vec::with_capacity(3);
|
||||
if rt.kind != JsRuntimeKind::Deno {
|
||||
args.push(OsString::from("--no-js-runtimes"));
|
||||
}
|
||||
args.push(OsString::from("--js-runtimes"));
|
||||
args.push(rt.spec());
|
||||
args
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::downloader::ytdlp::STATE_DIR_ENV;
|
||||
use std::{fs, sync::MutexGuard};
|
||||
use tempfile::TempDir;
|
||||
|
||||
/// Serialises env-mutating tests and clears every var the resolver reads.
|
||||
fn env_guard() -> MutexGuard<'static, ()> {
|
||||
let guard = crate::downloader::RESOLVER_ENV_LOCK
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
for key in [JS_RUNTIME_ENV, DENO_ENV, STATE_DIR_ENV] {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
guard
|
||||
}
|
||||
|
||||
/// Points the state dir at an (initially empty) dir under `tmp` and returns it.
|
||||
fn set_state(tmp: &TempDir) -> PathBuf {
|
||||
let state = tmp.path().join("state");
|
||||
fs::create_dir_all(&state).unwrap();
|
||||
unsafe { env::set_var(STATE_DIR_ENV, &state) };
|
||||
state
|
||||
}
|
||||
|
||||
/// Writes a fake `deno` script to `path` (every caller uses a fresh path), then waits until
|
||||
/// it can be exec'd. A child forked by a parallel test while our write fd was open keeps a
|
||||
/// copy of it until that child execs, so our exec can fail with ETXTBSY
|
||||
/// (rust-lang/rust#114554). One successful exec proves no writer is left, and none can
|
||||
/// appear later because our fd is already closed.
|
||||
fn fake_deno(path: &Path, ver: &str) {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
fs::create_dir_all(path.parent().unwrap()).unwrap();
|
||||
fs::write(
|
||||
path,
|
||||
format!("#!/bin/sh\necho 'deno {ver} (stable, release, test)'\necho 'v8 1.0'\n"),
|
||||
)
|
||||
.unwrap();
|
||||
fs::set_permissions(path, fs::Permissions::from_mode(0o755)).unwrap();
|
||||
for _ in 0..200 {
|
||||
match Command::new(path).arg("--version").output() {
|
||||
Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy => {
|
||||
std::thread::sleep(std::time::Duration::from_millis(5));
|
||||
}
|
||||
_ => return,
|
||||
}
|
||||
}
|
||||
panic!("{} stayed busy (ETXTBSY)", path.display());
|
||||
}
|
||||
|
||||
fn deno_at(role: JsRuntimeRole, p: PathBuf) -> Option<(JsRuntimeRole, JsRuntime)> {
|
||||
Some((role, JsRuntime { kind: JsRuntimeKind::Deno, path: Some(p) }))
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spec_parsing_accepts_allowlisted_runtimes() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let file = tmp.path().join("node");
|
||||
fs::write(&file, "").unwrap();
|
||||
|
||||
assert_eq!(
|
||||
parse_js_runtime_spec("deno"),
|
||||
Ok(JsRuntime { kind: JsRuntimeKind::Deno, path: None })
|
||||
);
|
||||
assert_eq!(parse_js_runtime_spec(" NODE ").unwrap().kind, JsRuntimeKind::Node);
|
||||
assert_eq!(
|
||||
parse_js_runtime_spec(&format!("node:{}", file.display())),
|
||||
Ok(JsRuntime { kind: JsRuntimeKind::Node, path: Some(file.clone()) })
|
||||
);
|
||||
assert_eq!(
|
||||
parse_js_runtime_spec(&format!("deno:{}", tmp.path().display())).unwrap().path,
|
||||
Some(tmp.path().to_path_buf())
|
||||
);
|
||||
assert_eq!(parse_js_runtime_spec("bun").unwrap().kind, JsRuntimeKind::Bun);
|
||||
assert_eq!(parse_js_runtime_spec("quickjs").unwrap().kind, JsRuntimeKind::QuickJs);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spec_parsing_rejects_bad_values() {
|
||||
assert_eq!(parse_js_runtime_spec("node:"), Err("empty path after ':'".into()));
|
||||
assert_eq!(parse_js_runtime_spec("node:rel/path"), Err("path must be absolute".into()));
|
||||
assert_eq!(
|
||||
parse_js_runtime_spec("deno:/does/not/exist"),
|
||||
Err("path does not exist".into())
|
||||
);
|
||||
for bad in ["python", "--exec", "deno,node"] {
|
||||
let err = parse_js_runtime_spec(bad).unwrap_err();
|
||||
assert!(err.starts_with("unknown runtime"), "{bad}: {err}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn args_follow_runtime_kind() {
|
||||
assert!(js_runtime_args(None).is_empty());
|
||||
let deno = JsRuntime { kind: JsRuntimeKind::Deno, path: Some("/p".into()) };
|
||||
assert_eq!(js_runtime_args(Some(&deno)), ["--js-runtimes", "deno:/p"]);
|
||||
let node = JsRuntime { kind: JsRuntimeKind::Node, path: Some("/p".into()) };
|
||||
assert_eq!(
|
||||
js_runtime_args(Some(&node)),
|
||||
["--no-js-runtimes", "--js-runtimes", "node:/p"]
|
||||
);
|
||||
let bun = JsRuntime { kind: JsRuntimeKind::Bun, path: None };
|
||||
assert_eq!(js_runtime_args(Some(&bun)), ["--no-js-runtimes", "--js-runtimes", "bun"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn version_parsing() {
|
||||
let v297 = Some(DenoVersion { major: 2, minor: 9, patch: 7 });
|
||||
assert_eq!(
|
||||
parse_deno_version_output(
|
||||
"deno 2.9.7 (stable, release, aarch64-apple-darwin)\nv8 14.0\ntypescript 5.9\n"
|
||||
),
|
||||
v297
|
||||
);
|
||||
assert_eq!(parse_deno_version_output("deno 2.9.7+abc123 (canary, x)\n"), v297);
|
||||
assert_eq!(parse_deno_version_output("node v22"), None);
|
||||
assert_eq!(parse_deno_version_output(""), None);
|
||||
assert_eq!(parse_deno_version_output("deno"), None);
|
||||
assert_eq!(DenoVersion::parse("v2.9.7"), v297);
|
||||
assert_eq!(DenoVersion::parse("2.9"), None);
|
||||
assert_eq!(DenoVersion::parse("2.9.7.1"), None);
|
||||
assert!(DenoVersion::parse("2.10.0") > v297);
|
||||
assert_eq!(v297.unwrap().to_string(), "2.9.7");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn newer_state_dir_deno_wins() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.4.0");
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn newer_pinned_deno_wins() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.10.0");
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
deno_at(JsRuntimeRole::Pinned, pinned)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tie_goes_to_state_dir() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.9.7");
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn too_old_pinned_falls_back_to_path() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.2.9");
|
||||
let bin = tmp.path().join("bin");
|
||||
fake_deno(&bin.join("deno"), "2.4.0");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(Some(bin.as_os_str())),
|
||||
deno_at(JsRuntimeRole::Path, bin.join("deno"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn valid_pinned_beats_newer_path_deno() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.4.0");
|
||||
let bin = tmp.path().join("bin");
|
||||
fake_deno(&bin.join("deno"), "2.9.7");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(Some(bin.as_os_str())),
|
||||
deno_at(JsRuntimeRole::Pinned, pinned)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pinned_deno_also_on_path_wins_as_pinned_only() {
|
||||
// The Nix wrappers set ARCHIVR_DENO and put the same Deno on PATH.
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let bin = tmp.path().join("bin");
|
||||
let deno = bin.join("deno");
|
||||
fake_deno(&deno, "2.9.4");
|
||||
unsafe { env::set_var(DENO_ENV, &deno) };
|
||||
assert_eq!(find_on_path("deno", Some(bin.as_os_str())), Some(deno.clone()));
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(Some(bin.as_os_str())),
|
||||
deno_at(JsRuntimeRole::Pinned, deno)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn forced_pinned_deno_wins_as_forced_only() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let bin = tmp.path().join("bin");
|
||||
let deno = bin.join("deno");
|
||||
fake_deno(&deno, "2.9.4");
|
||||
unsafe {
|
||||
env::set_var(DENO_ENV, &deno);
|
||||
env::set_var(JS_RUNTIME_ENV, format!("deno:{}", deno.display()));
|
||||
}
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(Some(bin.as_os_str())),
|
||||
deno_at(JsRuntimeRole::Forced, deno)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn too_old_path_deno_resolves_none() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let bin = tmp.path().join("bin");
|
||||
fake_deno(&bin.join("deno"), "2.2.9");
|
||||
assert_eq!(resolve_js_runtime_with_path(Some(bin.as_os_str())), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn valid_forced_runtime_wins() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.9.7");
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
let node = tmp.path().join("node");
|
||||
fs::write(&node, "").unwrap();
|
||||
unsafe {
|
||||
env::set_var(DENO_ENV, &pinned);
|
||||
env::set_var(JS_RUNTIME_ENV, format!("node:{}", node.display()));
|
||||
}
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
Some((
|
||||
JsRuntimeRole::Forced,
|
||||
JsRuntime { kind: JsRuntimeKind::Node, path: Some(node) }
|
||||
))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_forced_runtime_is_ignored() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
unsafe { env::set_var(JS_RUNTIME_ENV, "python") };
|
||||
assert!(forced_js_runtime().is_err());
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_candidates_resolves_none() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let empty_bin = tmp.path().join("bin");
|
||||
fs::create_dir_all(&empty_bin).unwrap();
|
||||
assert!(deno_candidates().is_empty());
|
||||
assert_eq!(resolve_js_runtime_with_path(Some(empty_bin.as_os_str())), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn forced_directory_probe_joins_runtime_name() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
fake_deno(&tmp.path().join("deno"), "2.9.7");
|
||||
let rt = JsRuntime { kind: JsRuntimeKind::Deno, path: Some(tmp.path().to_path_buf()) };
|
||||
assert_eq!(probe_js_runtime_version(&rt).as_deref(), Some("2.9.7"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn find_on_path_scans_in_order() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let (a, b) = (tmp.path().join("a"), tmp.path().join("b"));
|
||||
fs::create_dir_all(&a).unwrap();
|
||||
fs::create_dir_all(&b).unwrap();
|
||||
fs::write(b.join("tool"), "").unwrap();
|
||||
let path_var = env::join_paths([&a, &b]).unwrap();
|
||||
assert_eq!(find_on_path("tool", Some(&path_var)), Some(b.join("tool")));
|
||||
assert_eq!(find_on_path("missing", Some(&path_var)), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn refresh_js_runtime_swaps_cached_choice() {
|
||||
let _g = env_guard();
|
||||
unsafe { env::set_var(JS_RUNTIME_ENV, "node") };
|
||||
assert_eq!(refresh_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
|
||||
assert_eq!(resolve_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
|
||||
|
||||
unsafe { env::set_var(JS_RUNTIME_ENV, "bun") };
|
||||
assert_eq!(resolve_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
|
||||
assert_eq!(refresh_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Bun));
|
||||
|
||||
unsafe { env::remove_var(JS_RUNTIME_ENV) };
|
||||
refresh_js_runtime();
|
||||
}
|
||||
}
|
||||
|
|
@ -8,3 +8,32 @@ pub mod http;
|
|||
pub mod singlefile;
|
||||
pub mod font_extractor;
|
||||
pub mod text;
|
||||
pub mod js_runtime;
|
||||
pub mod deno_install;
|
||||
pub mod ytdlp_tools;
|
||||
|
||||
/// Env vars are process-global; every core test that sets resolver env vars takes this lock.
|
||||
#[cfg(test)]
|
||||
pub(crate) static RESOLVER_ENV_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||
|
||||
/// Writes an executable script to `path` (callers must use a fresh path each time), then
|
||||
/// waits until it can be exec'd. A child forked by a parallel test while our write fd was
|
||||
/// open keeps a copy of it until that child execs, so our own exec can fail with ETXTBSY
|
||||
/// (rust-lang/rust#114554). One exec that isn't ETXTBSY proves no writer is left, and none
|
||||
/// can appear later because our fd is already closed.
|
||||
#[cfg(all(test, unix))]
|
||||
pub(crate) fn write_script(path: &std::path::Path, body: &str) {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
|
||||
std::fs::write(path, body).unwrap();
|
||||
std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o755)).unwrap();
|
||||
for _ in 0..200 {
|
||||
match std::process::Command::new(path).arg("--version").output() {
|
||||
Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy => {
|
||||
std::thread::sleep(std::time::Duration::from_millis(5));
|
||||
}
|
||||
_ => return,
|
||||
}
|
||||
}
|
||||
panic!("{} stayed busy (ETXTBSY)", path.display());
|
||||
}
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
559
crates/archivr-core/src/downloader/ytdlp_tools.rs
Normal file
559
crates/archivr-core/src/downloader/ytdlp_tools.rs
Normal file
|
|
@ -0,0 +1,559 @@
|
|||
//! yt-dlp + Deno self-update and status shared by `archivr yt-dlp` and the admin API.
|
||||
//! Sync; network via blocking reqwest.
|
||||
|
||||
use anyhow::{anyhow, bail, Context, Result};
|
||||
use serde::Serialize;
|
||||
use std::{
|
||||
env,
|
||||
path::{Path, PathBuf},
|
||||
process::Command,
|
||||
};
|
||||
|
||||
use super::deno_install;
|
||||
use super::js_runtime::{
|
||||
forced_js_runtime, path_deno, pinned_deno, probe_deno_version, probe_js_runtime_version,
|
||||
refresh_js_runtime, resolve_js_runtime_with_role, state_dir_deno, JsRuntimeRole,
|
||||
JS_RUNTIME_ENV,
|
||||
};
|
||||
use super::ytdlp::{
|
||||
forced_yt_dlp, pinned_yt_dlp, probe_version, refresh_yt_dlp, resolve_yt_dlp, state_dir,
|
||||
state_dir_yt_dlp,
|
||||
};
|
||||
|
||||
/// GitHub release metadata endpoint for the upstream yt-dlp project.
|
||||
pub const YT_DLP_LATEST_RELEASE: &str =
|
||||
"https://api.github.com/repos/yt-dlp/yt-dlp/releases/latest";
|
||||
|
||||
/// Every python zipapp starts with this shebang; used as a sanity check that we
|
||||
/// downloaded the artifact and not an HTML error page or an LFS pointer.
|
||||
const ZIPAPP_SHEBANG: &[u8] = b"#!/usr/bin/env python3";
|
||||
|
||||
/// One candidate slot of a `status` table.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct ToolCandidate {
|
||||
/// Stable key: "force" | "env" | "state-dir" | "path".
|
||||
pub role: &'static str,
|
||||
/// Exact CLI row label.
|
||||
pub label: &'static str,
|
||||
/// `None` = empty slot (the CLI renders dashes).
|
||||
pub path: Option<String>,
|
||||
pub version: Option<String>,
|
||||
pub chosen: bool,
|
||||
/// Why the candidate can't be used: an invalid `ARCHIVR_JS_RUNTIME`, or a yt-dlp
|
||||
/// candidate that exists but whose `--version` probe fails (last stderr line).
|
||||
pub invalid: Option<String>,
|
||||
}
|
||||
|
||||
/// The candidate the resolver picked.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct ChosenTool {
|
||||
/// `None` if the cached yt-dlp path matches no row.
|
||||
pub role: Option<&'static str>,
|
||||
/// JS only: "deno" | "node" | "bun" | "quickjs".
|
||||
pub kind: Option<&'static str>,
|
||||
pub path: Option<String>,
|
||||
pub version: Option<String>,
|
||||
}
|
||||
|
||||
/// Everything `archivr yt-dlp status` prints, as data.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct ToolsStatus {
|
||||
/// force, env, state-dir, path-fallback (CLI order).
|
||||
pub yt_dlp: Vec<ToolCandidate>,
|
||||
pub yt_dlp_chosen: ChosenTool,
|
||||
/// force, env (ARCHIVR_DENO), state-dir, path (deno).
|
||||
pub js_runtime: Vec<ToolCandidate>,
|
||||
pub js_runtime_chosen: Option<ChosenTool>,
|
||||
pub state_dir: Option<String>,
|
||||
pub yt_dlp_target: Option<String>,
|
||||
pub yt_dlp_installed: bool,
|
||||
pub deno_target: Option<String>,
|
||||
pub deno_installed: bool,
|
||||
}
|
||||
|
||||
/// Per-component outcome of [`update_tools`]; `Ok` carries a one-line human outcome.
|
||||
pub struct UpdateReport {
|
||||
pub yt_dlp: Result<String>,
|
||||
pub deno: Result<String>,
|
||||
}
|
||||
|
||||
impl UpdateReport {
|
||||
/// Names of the failed components, in `["yt-dlp", "deno"]` order.
|
||||
pub fn failed_components(&self) -> Vec<&'static str> {
|
||||
[("yt-dlp", self.yt_dlp.is_err()), ("deno", self.deno.is_err())]
|
||||
.into_iter()
|
||||
.filter_map(|(name, failed)| failed.then_some(name))
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolves `<state_dir>/yt-dlp/`, erroring out if there is no usable HOME.
|
||||
fn yt_dlp_state_dir() -> Result<PathBuf> {
|
||||
state_dir()
|
||||
.map(|d| d.join("yt-dlp"))
|
||||
.context("could not determine a state directory (is $HOME set?)")
|
||||
}
|
||||
|
||||
/// Asks the GitHub API for the newest yt-dlp release tag.
|
||||
fn latest_yt_dlp_version(client: &reqwest::blocking::Client) -> Result<String> {
|
||||
let body = client
|
||||
.get(YT_DLP_LATEST_RELEASE)
|
||||
.send()
|
||||
.context("failed to reach the GitHub releases API")?
|
||||
.error_for_status()
|
||||
.context("GitHub releases API returned an error")?
|
||||
.text()
|
||||
.context("failed to read the GitHub releases API response")?;
|
||||
|
||||
let json: serde_json::Value =
|
||||
serde_json::from_str(&body).context("GitHub releases API returned invalid JSON")?;
|
||||
|
||||
json.get("tag_name")
|
||||
.and_then(|t| t.as_str())
|
||||
.map(str::to_string)
|
||||
.context("GitHub releases API response had no tag_name")
|
||||
}
|
||||
|
||||
/// Installs or updates the yt-dlp zipapp in the state dir. Progress goes to `log`;
|
||||
/// `Ok` carries a one-line human outcome.
|
||||
pub fn install_yt_dlp(
|
||||
client: &reqwest::blocking::Client,
|
||||
requested_version: Option<&str>,
|
||||
log: &mut dyn FnMut(&str),
|
||||
) -> Result<String> {
|
||||
let dir = yt_dlp_state_dir()?;
|
||||
let target = dir.join("yt-dlp");
|
||||
let staging = dir.join("yt-dlp.new");
|
||||
let version_file = dir.join(".version");
|
||||
|
||||
let version = match requested_version {
|
||||
Some(v) => v.to_string(),
|
||||
None => latest_yt_dlp_version(client)?,
|
||||
};
|
||||
|
||||
// The sibling .version file is what lets us skip a ~3MB download on a
|
||||
// no-op update; the binary itself is a zipapp with no cheap version probe
|
||||
// that doesn't cost a python startup.
|
||||
let installed = std::fs::read_to_string(&version_file).ok();
|
||||
if target.is_file() && installed.as_deref().map(str::trim) == Some(version.as_str()) {
|
||||
log(&format!("yt-dlp {version} is already installed at {}", target.display()));
|
||||
return Ok(format!("yt-dlp {version} already installed at {}", target.display()));
|
||||
}
|
||||
|
||||
log(&format!("Downloading yt-dlp {version}…"));
|
||||
let url = format!("https://github.com/yt-dlp/yt-dlp/releases/download/{version}/yt-dlp");
|
||||
let bytes = client
|
||||
.get(&url)
|
||||
.send()
|
||||
.with_context(|| format!("failed to download {url}"))?
|
||||
.error_for_status()
|
||||
.with_context(|| format!("download failed — is {version} a real release tag?"))?
|
||||
.bytes()
|
||||
.context("failed to read the downloaded yt-dlp body")?;
|
||||
|
||||
if !bytes.starts_with(ZIPAPP_SHEBANG) {
|
||||
bail!(
|
||||
"downloaded artifact from {url} is not a python zipapp \
|
||||
(expected it to start with `{}`) — refusing to install it",
|
||||
String::from_utf8_lossy(ZIPAPP_SHEBANG)
|
||||
);
|
||||
}
|
||||
|
||||
std::fs::create_dir_all(&dir)
|
||||
.with_context(|| format!("failed to create {}", dir.display()))?;
|
||||
std::fs::write(&staging, &bytes)
|
||||
.with_context(|| format!("failed to write {}", staging.display()))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
std::fs::set_permissions(&staging, std::fs::Permissions::from_mode(0o755))
|
||||
.with_context(|| format!("failed to chmod +x {}", staging.display()))?;
|
||||
}
|
||||
|
||||
// Atomic swap: a concurrently-running archivr sees either the whole old
|
||||
// binary or the whole new one, never a half-written file.
|
||||
std::fs::rename(&staging, &target)
|
||||
.with_context(|| format!("failed to install {}", target.display()))?;
|
||||
std::fs::write(&version_file, format!("{version}\n"))
|
||||
.with_context(|| format!("failed to record version in {}", version_file.display()))?;
|
||||
|
||||
// The zipapp is python source, not a native binary — installing it on a
|
||||
// host without python3 is legal (the server may run under a nix wrapper
|
||||
// with its own PATH) but worth flagging loudly. Kept as `warning:` (not
|
||||
// `warn:`) so the CLI's stderr is unchanged.
|
||||
let has_python = Command::new("python3")
|
||||
.arg("--version")
|
||||
.output()
|
||||
.map(|o| o.status.success())
|
||||
.unwrap_or(false);
|
||||
if !has_python {
|
||||
eprintln!(
|
||||
"warning: python3 was not found on PATH — the yt-dlp zipapp just installed \
|
||||
at {} will not run until python3 is available",
|
||||
target.display()
|
||||
);
|
||||
}
|
||||
let python_note = (!has_python)
|
||||
.then_some("; warning: python3 not found on PATH — the zipapp will not run until it is");
|
||||
|
||||
log(&format!("Installed yt-dlp {version} to {}", target.display()));
|
||||
log("archivr will now prefer it whenever it is newer than the pinned binary (ARCHIVR_YT_DLP).");
|
||||
|
||||
Ok(format!(
|
||||
"installed yt-dlp {version} to {}{}",
|
||||
target.display(),
|
||||
python_note.unwrap_or("")
|
||||
))
|
||||
}
|
||||
|
||||
/// Hint appended when the installed zipapp does not run; the zipapp is python source.
|
||||
const PYTHON_HINT: &str = "yt-dlp needs Python ≥ 3.10 on the server's PATH as `python3`";
|
||||
|
||||
/// Installs yt-dlp and Deno independently: a Deno failure never blocks the yt-dlp
|
||||
/// update (and vice versa). `Err` only if the HTTP client cannot be built.
|
||||
///
|
||||
/// With `refresh` (long-running server), the installed yt-dlp is probed with
|
||||
/// `--version` — an install that does not run is reported as a failure naming the
|
||||
/// cause — and each successful component refreshes its resolver cache, so the next
|
||||
/// yt-dlp call uses the new binary. The one-shot CLI passes `false`: it has no cache
|
||||
/// worth refreshing, and its output and probe count stay as before.
|
||||
pub fn update_tools(
|
||||
requested_yt_dlp_version: Option<&str>,
|
||||
user_agent: &str,
|
||||
refresh: bool,
|
||||
log: &mut dyn FnMut(&str),
|
||||
) -> Result<UpdateReport> {
|
||||
let client = reqwest::blocking::Client::builder()
|
||||
.user_agent(user_agent)
|
||||
.build()
|
||||
.context("failed to build an HTTP client")?;
|
||||
|
||||
let mut yt_dlp = install_yt_dlp(&client, requested_yt_dlp_version, log);
|
||||
if refresh && yt_dlp.is_ok() {
|
||||
if let Some(target) = state_dir_yt_dlp() {
|
||||
if let Err(Some(reason)) = probe_version_detail(&target) {
|
||||
yt_dlp = Err(anyhow!(unusable_install_message(&target, &reason)));
|
||||
}
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
}
|
||||
let deno = deno_install::install_deno(&client, log);
|
||||
if refresh && deno.is_ok() {
|
||||
refresh_js_runtime();
|
||||
}
|
||||
Ok(UpdateReport { yt_dlp, deno })
|
||||
}
|
||||
|
||||
/// Error text for an installed yt-dlp whose `--version` probe failed.
|
||||
fn unusable_install_message(target: &Path, reason: &str) -> String {
|
||||
format!(
|
||||
"installed {} but it does not run: {reason} — {PYTHON_HINT}",
|
||||
target.display()
|
||||
)
|
||||
}
|
||||
|
||||
/// Short reason for a failed `--version` run: the last non-empty stderr line (a Python
|
||||
/// traceback ends with the actual error), else the exit status.
|
||||
fn probe_failure_reason(stderr: &[u8], status: &str) -> String {
|
||||
String::from_utf8_lossy(stderr)
|
||||
.lines()
|
||||
.map(str::trim)
|
||||
.rfind(|l| !l.is_empty())
|
||||
.map_or_else(|| format!("--version failed ({status})"), str::to_string)
|
||||
}
|
||||
|
||||
/// Runs `<binary> --version`. `Err(None)` = nothing to run (not found); `Err(Some(reason))`
|
||||
/// = the binary exists but the probe failed.
|
||||
fn probe_version_detail(binary: &Path) -> std::result::Result<String, Option<String>> {
|
||||
let out = match Command::new(binary).arg("--version").output() {
|
||||
Ok(out) => out,
|
||||
Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Err(None),
|
||||
Err(e) => return Err(Some(format!("could not run: {e}"))),
|
||||
};
|
||||
if !out.status.success() {
|
||||
return Err(Some(probe_failure_reason(&out.stderr, &out.status.to_string())));
|
||||
}
|
||||
let version = String::from_utf8_lossy(&out.stdout).trim().to_string();
|
||||
if version.is_empty() {
|
||||
return Err(Some("--version printed nothing".into()));
|
||||
}
|
||||
Ok(version)
|
||||
}
|
||||
|
||||
fn display(p: &Path) -> String {
|
||||
p.display().to_string()
|
||||
}
|
||||
|
||||
/// One yt-dlp row, probing the candidate's version; a failing probe sets `invalid`.
|
||||
fn yt_row(role: &'static str, label: &'static str, path: Option<&Path>, chosen: &Path) -> ToolCandidate {
|
||||
let (version, invalid) = match path.map(probe_version_detail) {
|
||||
Some(Ok(v)) => (Some(v), None),
|
||||
Some(Err(reason)) => (None, reason),
|
||||
None => (None, None),
|
||||
};
|
||||
ToolCandidate {
|
||||
role,
|
||||
label,
|
||||
path: path.map(display),
|
||||
version,
|
||||
chosen: path == Some(chosen),
|
||||
invalid,
|
||||
}
|
||||
}
|
||||
|
||||
/// Every yt-dlp and JS runtime candidate, its version, and which one wins — the data
|
||||
/// `archivr yt-dlp status` prints. Spawns `--version` probes; call off async threads.
|
||||
pub fn tools_status() -> ToolsStatus {
|
||||
// yt-dlp: the cached choice, i.e. what this process actually runs.
|
||||
let chosen = resolve_yt_dlp();
|
||||
let state_candidate = state_dir_yt_dlp().filter(|p| p.is_file());
|
||||
let yt_dlp = vec![
|
||||
yt_row("force", "force (ARCHIVR_YT_DLP_FORCE)", forced_yt_dlp().as_deref(), &chosen),
|
||||
yt_row("env", "env (ARCHIVR_YT_DLP)", pinned_yt_dlp().as_deref(), &chosen),
|
||||
yt_row("state-dir", "state-dir", state_candidate.as_deref(), &chosen),
|
||||
yt_row("path", "path-fallback (yt-dlp)", Some(Path::new("yt-dlp")), &chosen),
|
||||
];
|
||||
let yt_dlp_chosen = match yt_dlp.iter().find(|c| c.chosen) {
|
||||
Some(c) => ChosenTool {
|
||||
role: Some(c.role),
|
||||
kind: None,
|
||||
path: Some(display(&chosen)),
|
||||
version: c.version.clone(),
|
||||
},
|
||||
None => ChosenTool {
|
||||
role: None,
|
||||
kind: None,
|
||||
path: Some(display(&chosen)),
|
||||
version: probe_version(&chosen),
|
||||
},
|
||||
};
|
||||
|
||||
// JS runtime: uncached and silent, so status never prints the resolver warnings.
|
||||
let js_chosen = resolve_js_runtime_with_role();
|
||||
let chosen_role = js_chosen.as_ref().map(|(role, _)| *role);
|
||||
let force_role = JsRuntimeRole::Forced;
|
||||
let force_row = match forced_js_runtime() {
|
||||
Ok(Some(rt)) => ToolCandidate {
|
||||
role: force_role.key(),
|
||||
label: force_role.label(),
|
||||
path: Some(rt.spec().to_string_lossy().into_owned()),
|
||||
version: probe_js_runtime_version(&rt),
|
||||
chosen: chosen_role == Some(force_role),
|
||||
invalid: None,
|
||||
},
|
||||
Ok(None) => ToolCandidate {
|
||||
role: force_role.key(),
|
||||
label: force_role.label(),
|
||||
path: None,
|
||||
version: None,
|
||||
chosen: false,
|
||||
invalid: None,
|
||||
},
|
||||
Err(reason) => ToolCandidate {
|
||||
role: force_role.key(),
|
||||
label: force_role.label(),
|
||||
path: Some(
|
||||
env::var_os(JS_RUNTIME_ENV)
|
||||
.unwrap_or_default()
|
||||
.to_string_lossy()
|
||||
.into_owned(),
|
||||
),
|
||||
version: None,
|
||||
chosen: false,
|
||||
invalid: Some(reason),
|
||||
},
|
||||
};
|
||||
let deno_row = |role: JsRuntimeRole, path: Option<PathBuf>| ToolCandidate {
|
||||
role: role.key(),
|
||||
label: role.label(),
|
||||
version: path
|
||||
.as_deref()
|
||||
.and_then(probe_deno_version)
|
||||
.map(|v| v.to_string()),
|
||||
path: path.as_deref().map(display),
|
||||
chosen: chosen_role == Some(role),
|
||||
invalid: None,
|
||||
};
|
||||
let js_runtime = vec![
|
||||
force_row,
|
||||
deno_row(JsRuntimeRole::Pinned, pinned_deno()),
|
||||
deno_row(JsRuntimeRole::StateDir, state_dir_deno().filter(|p| p.is_file())),
|
||||
deno_row(JsRuntimeRole::Path, path_deno()),
|
||||
];
|
||||
let js_runtime_chosen = js_chosen.map(|(role, rt)| ChosenTool {
|
||||
role: Some(role.key()),
|
||||
kind: Some(rt.kind.as_str()),
|
||||
path: rt.path.as_deref().map(display),
|
||||
version: js_runtime
|
||||
.iter()
|
||||
.find(|c| c.chosen)
|
||||
.and_then(|c| c.version.clone()),
|
||||
});
|
||||
|
||||
let yt_dlp_target = state_dir_yt_dlp();
|
||||
let deno_target = state_dir_deno();
|
||||
ToolsStatus {
|
||||
yt_dlp,
|
||||
yt_dlp_chosen,
|
||||
js_runtime,
|
||||
js_runtime_chosen,
|
||||
state_dir: state_dir().as_deref().map(display),
|
||||
yt_dlp_installed: yt_dlp_target.as_deref().is_some_and(Path::is_file),
|
||||
yt_dlp_target: yt_dlp_target.as_deref().map(display),
|
||||
deno_installed: deno_target.as_deref().is_some_and(Path::is_file),
|
||||
deno_target: deno_target.as_deref().map(display),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::downloader::js_runtime::DENO_ENV;
|
||||
use crate::downloader::ytdlp::{STATE_DIR_ENV, YT_DLP_ENV, YT_DLP_FORCE_ENV};
|
||||
use anyhow::anyhow;
|
||||
|
||||
const RESOLVER_ENVS: [&str; 5] =
|
||||
[YT_DLP_FORCE_ENV, YT_DLP_ENV, STATE_DIR_ENV, JS_RUNTIME_ENV, DENO_ENV];
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn tools_status_reports_forced_and_invalid_override() {
|
||||
let _guard = crate::downloader::RESOLVER_ENV_LOCK
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
for key in RESOLVER_ENVS {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let state = tmp.path().join("state");
|
||||
let forced = tmp.path().join("forced/yt-dlp");
|
||||
crate::downloader::write_script(&forced, "#!/bin/sh\necho 2020.01.01\n");
|
||||
unsafe {
|
||||
env::set_var(STATE_DIR_ENV, &state);
|
||||
env::set_var(YT_DLP_FORCE_ENV, &forced);
|
||||
env::set_var(JS_RUNTIME_ENV, "python");
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
|
||||
let status = tools_status();
|
||||
assert_eq!(status.yt_dlp[0].role, "force");
|
||||
assert!(status.yt_dlp[0].chosen);
|
||||
assert_eq!(status.yt_dlp[0].version.as_deref(), Some("2020.01.01"));
|
||||
assert_eq!(status.yt_dlp_chosen.role, Some("force"));
|
||||
let js_force = &status.js_runtime[0];
|
||||
assert!(
|
||||
js_force
|
||||
.invalid
|
||||
.as_deref()
|
||||
.is_some_and(|r| r.contains("unknown runtime python")),
|
||||
"{js_force:?}"
|
||||
);
|
||||
assert!(!js_force.chosen);
|
||||
assert!(!status.yt_dlp_installed);
|
||||
assert!(status
|
||||
.yt_dlp_target
|
||||
.as_deref()
|
||||
.is_some_and(|t| t.ends_with("yt-dlp/yt-dlp")));
|
||||
let json = serde_json::to_value(&status).unwrap();
|
||||
for key in ["yt_dlp", "js_runtime", "state_dir", "deno_target"] {
|
||||
assert!(json.get(key).is_some(), "missing {key}");
|
||||
}
|
||||
|
||||
for key in RESOLVER_ENVS {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn probe_failure_reason_prefers_last_stderr_line() {
|
||||
assert_eq!(
|
||||
probe_failure_reason(
|
||||
b"Traceback (most recent call last):\n File \"yt_dlp/__main__.py\", line 13\n\
|
||||
ImportError: You are using an unsupported version of Python. Only Python \
|
||||
versions 3.10 and above are supported by yt-dlp\n\n",
|
||||
"exit status: 1"
|
||||
),
|
||||
"ImportError: You are using an unsupported version of Python. Only Python \
|
||||
versions 3.10 and above are supported by yt-dlp"
|
||||
);
|
||||
assert_eq!(
|
||||
probe_failure_reason(b"\n boom: too old \n", "exit status: 1"),
|
||||
"boom: too old"
|
||||
);
|
||||
assert_eq!(
|
||||
probe_failure_reason(b" \n", "exit status: 2"),
|
||||
"--version failed (exit status: 2)"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unusable_install_message_names_cause_and_hint() {
|
||||
let msg = unusable_install_message(Path::new("/s/yt-dlp/yt-dlp"), "Only Python 3.10+");
|
||||
assert!(msg.contains("/s/yt-dlp/yt-dlp"), "{msg}");
|
||||
assert!(msg.contains("Only Python 3.10+"), "{msg}");
|
||||
assert!(msg.contains("Python ≥ 3.10"), "{msg}");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn probe_version_detail_classifies_outcomes() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let ok = tmp.path().join("ok");
|
||||
crate::downloader::write_script(&ok, "#!/bin/sh\necho 2024.01.01\n");
|
||||
assert_eq!(probe_version_detail(&ok), Ok("2024.01.01".into()));
|
||||
let bad = tmp.path().join("bad");
|
||||
crate::downloader::write_script(
|
||||
&bad,
|
||||
"#!/bin/sh\necho 'Only Python versions 3.10 and above are supported' >&2\nexit 1\n",
|
||||
);
|
||||
assert_eq!(
|
||||
probe_version_detail(&bad),
|
||||
Err(Some("Only Python versions 3.10 and above are supported".into()))
|
||||
);
|
||||
assert_eq!(probe_version_detail(&tmp.path().join("missing")), Err(None));
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn tools_status_reports_unusable_candidate_reason() {
|
||||
let _guard = crate::downloader::RESOLVER_ENV_LOCK
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
for key in RESOLVER_ENVS {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let pinned = tmp.path().join("pinned/yt-dlp");
|
||||
crate::downloader::write_script(&pinned, "#!/bin/sh\necho 'boom: too old' >&2\nexit 1\n");
|
||||
unsafe {
|
||||
env::set_var(STATE_DIR_ENV, tmp.path().join("state"));
|
||||
env::set_var(YT_DLP_ENV, &pinned);
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
|
||||
let status = tools_status();
|
||||
let env_row = &status.yt_dlp[1];
|
||||
assert_eq!(env_row.role, "env");
|
||||
assert_eq!(env_row.version, None);
|
||||
assert_eq!(env_row.invalid.as_deref(), Some("boom: too old"));
|
||||
|
||||
for key in RESOLVER_ENVS {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn failed_components_lists_only_failures() {
|
||||
let report = |y: bool, d: bool| UpdateReport {
|
||||
yt_dlp: if y { Ok("ok".into()) } else { Err(anyhow!("boom")) },
|
||||
deno: if d { Ok("ok".into()) } else { Err(anyhow!("boom")) },
|
||||
};
|
||||
assert!(report(true, true).failed_components().is_empty());
|
||||
assert_eq!(report(false, true).failed_components(), ["yt-dlp"]);
|
||||
assert_eq!(report(true, false).failed_components(), ["deno"]);
|
||||
assert_eq!(report(false, false).failed_components(), ["yt-dlp", "deno"]);
|
||||
}
|
||||
}
|
||||
109
crates/archivr-core/src/env_config.rs
Normal file
109
crates/archivr-core/src/env_config.rs
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
//! Env-var resolution helpers shared by the summary providers and the local
|
||||
//! transcription engines. External tools are configured by `ARCHIVR_*` env
|
||||
//! vars only, never TOML.
|
||||
|
||||
use anyhow::{Result, bail};
|
||||
use std::{
|
||||
env,
|
||||
path::{Path, PathBuf},
|
||||
};
|
||||
|
||||
/// Reads a required env var, failing with the *exact variable name* so the
|
||||
/// server can hand a caller an actionable 400 rather than "not configured".
|
||||
pub(crate) fn required_env(name: &str) -> Result<String> {
|
||||
match env::var(name) {
|
||||
Ok(v) if !v.trim().is_empty() => Ok(v),
|
||||
_ => bail!("missing required environment variable: {name}"),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn env_or(name: &str, default: &str) -> String {
|
||||
env::var(name)
|
||||
.ok()
|
||||
.filter(|v| !v.trim().is_empty())
|
||||
.unwrap_or_else(|| default.to_string())
|
||||
}
|
||||
|
||||
pub(crate) fn optional_env(name: &str) -> Option<String> {
|
||||
env::var(name).ok().filter(|v| !v.trim().is_empty())
|
||||
}
|
||||
|
||||
pub(crate) fn env_timeout(name: &str, default: u64) -> u64 {
|
||||
env::var(name)
|
||||
.ok()
|
||||
.and_then(|v| v.trim().parse::<u64>().ok())
|
||||
.filter(|v| *v > 0)
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
/// Resolve a CLI executable path.
|
||||
///
|
||||
/// Priority: `env_name` override → first `well_known_absolute` path that
|
||||
/// exists → `HOME/.local/bin/<bare>` if it exists → bare name (relies on the
|
||||
/// server's PATH). The macOS defaults matter for `codex`, which the ChatGPT
|
||||
/// desktop app installs at `/Applications/ChatGPT.app/Contents/Resources/codex`
|
||||
/// and does not add to PATH.
|
||||
pub(crate) fn resolve_cli(env_name: &str, well_known_absolute: &[&str], bare: &str) -> PathBuf {
|
||||
if let Some(explicit) = optional_env(env_name) {
|
||||
return PathBuf::from(explicit);
|
||||
}
|
||||
for candidate in well_known_absolute {
|
||||
let p = Path::new(candidate);
|
||||
if p.is_file() {
|
||||
return p.to_path_buf();
|
||||
}
|
||||
}
|
||||
if let Some(home) = env::var_os("HOME") {
|
||||
let mut p = PathBuf::from(home);
|
||||
p.push(".local/bin");
|
||||
p.push(bare);
|
||||
if p.is_file() {
|
||||
return p;
|
||||
}
|
||||
}
|
||||
PathBuf::from(bare)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
static ENV_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||
const VAR: &str = "ARCHIVR_TEST_RESOLVE_CLI";
|
||||
|
||||
#[test]
|
||||
fn resolve_cli_prefers_env_then_absolute_then_bare() {
|
||||
let _guard = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let absolute = dir.path().join("tool");
|
||||
std::fs::write(&absolute, b"").unwrap();
|
||||
let absolute_str = absolute.to_str().unwrap();
|
||||
let bare = "archivr-test-resolve-cli-surely-not-installed";
|
||||
|
||||
unsafe { env::set_var(VAR, "/explicit/tool") };
|
||||
assert_eq!(
|
||||
resolve_cli(VAR, &[absolute_str], bare),
|
||||
PathBuf::from("/explicit/tool")
|
||||
);
|
||||
|
||||
unsafe { env::remove_var(VAR) };
|
||||
assert_eq!(resolve_cli(VAR, &["/nonexistent/x", absolute_str], bare), absolute);
|
||||
assert_eq!(
|
||||
resolve_cli(VAR, &["/nonexistent/x"], bare),
|
||||
PathBuf::from(bare)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn env_timeout_ignores_zero_and_garbage() {
|
||||
let _guard = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
|
||||
const T: &str = "ARCHIVR_TEST_ENV_TIMEOUT";
|
||||
unsafe { env::set_var(T, "0") };
|
||||
assert_eq!(env_timeout(T, 7), 7);
|
||||
unsafe { env::set_var(T, "abc") };
|
||||
assert_eq!(env_timeout(T, 7), 7);
|
||||
unsafe { env::set_var(T, " 12 ") };
|
||||
assert_eq!(env_timeout(T, 7), 12);
|
||||
unsafe { env::remove_var(T) };
|
||||
}
|
||||
}
|
||||
|
|
@ -5,3 +5,8 @@ pub mod downloader;
|
|||
pub mod hash;
|
||||
pub mod twitter;
|
||||
pub mod summarizer;
|
||||
pub mod subtitles;
|
||||
pub mod thread_title;
|
||||
pub mod transcriber;
|
||||
pub(crate) mod env_config;
|
||||
pub(crate) mod process;
|
||||
|
|
|
|||
381
crates/archivr-core/src/process.rs
Normal file
381
crates/archivr-core/src/process.rs
Normal file
|
|
@ -0,0 +1,381 @@
|
|||
//! Subprocess runner with a wall-clock timeout.
|
||||
//!
|
||||
//! `archivr-core` deliberately has no async runtime and the tree carries no
|
||||
//! `wait_timeout` dependency, so the timeout is enforced by structure: stdout
|
||||
//! and stderr are drained on their own threads (a chatty child must never
|
||||
//! block on a full pipe buffer), stdin is written on a third thread (a large
|
||||
//! prompt can exceed the pipe buffer), and the calling thread polls
|
||||
//! `try_wait` until the child exits or the deadline passes, then kills it.
|
||||
|
||||
use anyhow::{Context, Result, anyhow};
|
||||
use std::{
|
||||
ffi::OsString,
|
||||
io::{Read, Write},
|
||||
path::Path,
|
||||
process::{Child, Command, Stdio},
|
||||
sync::mpsc,
|
||||
thread,
|
||||
time::{Duration, Instant},
|
||||
};
|
||||
|
||||
/// Bytes of stderr kept for diagnostics.
|
||||
const STDERR_TAIL_BYTES: usize = 4096;
|
||||
/// Characters of the stderr tail quoted in a non-zero-exit error.
|
||||
const EXIT_ERROR_STDERR_CHARS: usize = 400;
|
||||
const POLL_INTERVAL: Duration = Duration::from_millis(50);
|
||||
/// How long to wait for the pipe readers after the direct child exits before
|
||||
/// assuming a grandchild holds the pipes and killing the process group.
|
||||
pub(crate) const READER_GRACE: Duration = Duration::from_secs(2);
|
||||
|
||||
/// Puts the child in its own process group (unix) so a timeout can kill the
|
||||
/// whole tree, including grandchildren that inherited the output pipes.
|
||||
pub(crate) fn isolate_process_group(cmd: &mut Command) {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::process::CommandExt;
|
||||
cmd.process_group(0);
|
||||
}
|
||||
#[cfg(not(unix))]
|
||||
let _ = cmd;
|
||||
}
|
||||
|
||||
/// SIGKILLs the process group led by `pid` (spawned via
|
||||
/// [`isolate_process_group`]). Best effort; a missing group (`ESRCH`) is not
|
||||
/// an error. Calls `kill(2)` directly: slim runtime images ship no `kill`
|
||||
/// binary.
|
||||
pub(crate) fn kill_process_group(pid: u32) {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
// kill(0, ..) hits our own group and kill(-1, ..) every process we may
|
||||
// signal; a pid that doesn't fit pid_t can't be a real child either.
|
||||
let Ok(pgid) = libc::pid_t::try_from(pid) else {
|
||||
return;
|
||||
};
|
||||
if pgid <= 1 {
|
||||
return;
|
||||
}
|
||||
// SAFETY: kill(2) takes plain integers and touches no memory of ours;
|
||||
// a negative pid targets the process group `pgid`.
|
||||
let _ = unsafe { libc::kill(-pgid, libc::SIGKILL) };
|
||||
}
|
||||
#[cfg(not(unix))]
|
||||
let _ = pid;
|
||||
}
|
||||
|
||||
/// Kills the child's whole process group and reaps the direct child.
|
||||
pub(crate) fn kill_tree(child: &mut Child) {
|
||||
kill_process_group(child.id());
|
||||
let _ = child.kill();
|
||||
let _ = child.wait();
|
||||
}
|
||||
|
||||
/// Receives a reader result after the direct child exited: waits up to
|
||||
/// `min(grace, budget)`, then kills the process group (a grandchild holding
|
||||
/// the pipe) and waits one more grace period. `None` if still not done.
|
||||
pub(crate) fn recv_after_exit<T>(rx: &mpsc::Receiver<T>, pid: u32, budget: Duration) -> Option<T> {
|
||||
if let Ok(v) = rx.recv_timeout(READER_GRACE.min(budget)) {
|
||||
return Some(v);
|
||||
}
|
||||
kill_process_group(pid);
|
||||
rx.recv_timeout(READER_GRACE).ok()
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct ProcessOutput {
|
||||
pub stdout: String,
|
||||
/// Last 4 KiB of stderr, lossy UTF-8.
|
||||
#[allow(dead_code)]
|
||||
pub stderr_tail: String,
|
||||
}
|
||||
|
||||
/// Sentinel at the root of a timeout error, so callers can recognise a timeout
|
||||
/// without string matching (see [`is_process_timeout`]).
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct ProcessTimedOut {
|
||||
pub secs: u64,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for ProcessTimedOut {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "timed out after {}s", self.secs)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for ProcessTimedOut {}
|
||||
|
||||
/// True when `error` came from a [`run_with_timeout`] deadline, however many
|
||||
/// context layers have been added on top since.
|
||||
pub(crate) fn is_process_timeout(error: &anyhow::Error) -> bool {
|
||||
error
|
||||
.chain()
|
||||
.find_map(|c| c.downcast_ref::<ProcessTimedOut>())
|
||||
.or_else(|| error.downcast_ref::<ProcessTimedOut>())
|
||||
.is_some()
|
||||
}
|
||||
|
||||
/// Spawns `executable args…`, optionally writes `stdin`, drains stdout and
|
||||
/// stderr on their own threads, and kills the child if it is still running at
|
||||
/// `timeout`.
|
||||
///
|
||||
/// - Non-zero exit → `Err("{exe} exited with {status}: {last 400 chars of stderr}")`.
|
||||
/// - Timeout → an error whose root is [`ProcessTimedOut`] with the message
|
||||
/// `"{exe} timed out after {secs}s"`.
|
||||
pub(crate) fn run_with_timeout(
|
||||
executable: &Path,
|
||||
args: &[OsString],
|
||||
stdin: Option<&str>,
|
||||
timeout: Duration,
|
||||
) -> Result<ProcessOutput> {
|
||||
let exe = executable.display().to_string();
|
||||
let started = Instant::now();
|
||||
let timeout_error = || {
|
||||
let secs = timeout.as_secs().max(1);
|
||||
anyhow::Error::new(ProcessTimedOut { secs }).context(format!("{exe} timed out after {secs}s"))
|
||||
};
|
||||
|
||||
let mut command = Command::new(executable);
|
||||
command
|
||||
.args(args)
|
||||
.stdin(if stdin.is_some() {
|
||||
Stdio::piped()
|
||||
} else {
|
||||
Stdio::null()
|
||||
})
|
||||
.stdout(Stdio::piped())
|
||||
.stderr(Stdio::piped());
|
||||
isolate_process_group(&mut command);
|
||||
let mut child = command
|
||||
.spawn()
|
||||
.with_context(|| format!("failed to spawn {exe}"))?;
|
||||
let pid = child.id();
|
||||
|
||||
if let Some(input) = stdin {
|
||||
let mut pipe = child
|
||||
.stdin
|
||||
.take()
|
||||
.ok_or_else(|| anyhow!("failed to open stdin for {exe}"))?;
|
||||
let owned = input.to_string();
|
||||
thread::spawn(move || {
|
||||
let _ = pipe.write_all(owned.as_bytes());
|
||||
// Dropping the pipe closes it, which tells the child input is complete.
|
||||
});
|
||||
}
|
||||
|
||||
let mut stdout = child
|
||||
.stdout
|
||||
.take()
|
||||
.ok_or_else(|| anyhow!("failed to open stdout for {exe}"))?;
|
||||
let (out_tx, out_rx) = mpsc::channel();
|
||||
thread::spawn(move || {
|
||||
let mut buf = String::new();
|
||||
let res = stdout.read_to_string(&mut buf).map(|_| buf);
|
||||
let _ = out_tx.send(res);
|
||||
});
|
||||
|
||||
let mut stderr = child
|
||||
.stderr
|
||||
.take()
|
||||
.ok_or_else(|| anyhow!("failed to open stderr for {exe}"))?;
|
||||
let (err_tx, err_rx) = mpsc::channel();
|
||||
thread::spawn(move || {
|
||||
let mut tail: Vec<u8> = Vec::new();
|
||||
let mut chunk = [0u8; 8192];
|
||||
loop {
|
||||
match stderr.read(&mut chunk) {
|
||||
Ok(0) | Err(_) => break,
|
||||
Ok(n) => {
|
||||
tail.extend_from_slice(&chunk[..n]);
|
||||
if tail.len() > STDERR_TAIL_BYTES {
|
||||
let excess = tail.len() - STDERR_TAIL_BYTES;
|
||||
tail.drain(..excess);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
let _ = err_tx.send(String::from_utf8_lossy(&tail).into_owned());
|
||||
});
|
||||
|
||||
let status = loop {
|
||||
match child.try_wait() {
|
||||
Ok(Some(status)) => break status,
|
||||
Ok(None) => {}
|
||||
Err(e) => {
|
||||
kill_tree(&mut child);
|
||||
return Err(anyhow::Error::new(e).context(format!("failed to wait for {exe}")));
|
||||
}
|
||||
}
|
||||
if started.elapsed() >= timeout {
|
||||
// Kill the whole group so grandchildren release the pipes too.
|
||||
kill_tree(&mut child);
|
||||
return Err(timeout_error());
|
||||
}
|
||||
thread::sleep(POLL_INTERVAL);
|
||||
};
|
||||
|
||||
// The child has exited, but a grandchild that inherited the pipes can keep
|
||||
// them open; give the readers a short grace, then kill the group.
|
||||
let remaining = || timeout.saturating_sub(started.elapsed());
|
||||
let collected = match recv_after_exit(&out_rx, pid, remaining()) {
|
||||
Some(res) => res.with_context(|| format!("failed to read stdout of {exe}"))?,
|
||||
None => return Err(timeout_error()),
|
||||
};
|
||||
let Some(stderr_tail) = recv_after_exit(&err_rx, pid, remaining()) else {
|
||||
return Err(timeout_error());
|
||||
};
|
||||
|
||||
if !status.success() {
|
||||
anyhow::bail!(
|
||||
"{exe} exited with {status}: {}",
|
||||
last_chars(stderr_tail.trim(), EXIT_ERROR_STDERR_CHARS)
|
||||
);
|
||||
}
|
||||
Ok(ProcessOutput {
|
||||
stdout: collected,
|
||||
stderr_tail,
|
||||
})
|
||||
}
|
||||
|
||||
fn last_chars(s: &str, max: usize) -> String {
|
||||
let count = s.chars().count();
|
||||
if count <= max {
|
||||
return s.to_string();
|
||||
}
|
||||
let tail: String = s.chars().skip(count - max).collect();
|
||||
format!("…{tail}")
|
||||
}
|
||||
|
||||
/// Shared by process-group tests here and in `downloader::ytdlp`.
|
||||
#[cfg(all(test, unix))]
|
||||
pub(crate) mod test_support {
|
||||
use std::{path::Path, process::Command, time::{Duration, Instant}};
|
||||
|
||||
/// Shell snippet: start a background `sleep 30` and record its pid in `pid_file`.
|
||||
pub(crate) fn spawn_grandchild_snippet(pid_file: &Path) -> String {
|
||||
format!("sleep 30 & echo $! > '{}'; ", pid_file.display())
|
||||
}
|
||||
|
||||
/// Reads the pid written by [`spawn_grandchild_snippet`] and asserts the
|
||||
/// process disappears within a few seconds (allowing init to reap it).
|
||||
pub(crate) fn assert_grandchild_gone(pid_file: &Path) {
|
||||
let pid = std::fs::read_to_string(pid_file).unwrap().trim().to_string();
|
||||
assert!(!pid.is_empty(), "grandchild pid not recorded");
|
||||
let deadline = Instant::now() + Duration::from_secs(5);
|
||||
loop {
|
||||
let alive = Command::new("kill")
|
||||
.args(["-0", &pid])
|
||||
.stderr(std::process::Stdio::null())
|
||||
.status()
|
||||
.map(|s| s.success())
|
||||
.unwrap_or(false);
|
||||
if !alive {
|
||||
return;
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
let _ = Command::new("kill").args(["-KILL", &pid]).status();
|
||||
panic!("grandchild {pid} survived the timeout kill");
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(50));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn os(args: &[&str]) -> Vec<OsString> {
|
||||
args.iter().map(OsString::from).collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_with_timeout_drains_large_stderr_without_deadlock() {
|
||||
let out = run_with_timeout(
|
||||
Path::new("sh"),
|
||||
&os(&["-c", "head -c 1000000 /dev/zero | tr '\\0' x >&2; echo ok"]),
|
||||
None,
|
||||
Duration::from_secs(30),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(out.stdout, "ok\n");
|
||||
assert_eq!(out.stderr_tail.len(), STDERR_TAIL_BYTES);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_with_timeout_kills_overrunning_child_and_marks_timeout() {
|
||||
let started = Instant::now();
|
||||
let err = run_with_timeout(
|
||||
Path::new("sleep"),
|
||||
&os(&["30"]),
|
||||
None,
|
||||
Duration::from_secs(1),
|
||||
)
|
||||
.unwrap_err();
|
||||
assert!(is_process_timeout(&err), "{err:#}");
|
||||
assert!(format!("{err:#}").contains("timed out after 1s"), "{err:#}");
|
||||
assert!(started.elapsed() < Duration::from_secs(5));
|
||||
// Still recognisable under further context layers.
|
||||
let wrapped = err.context("outer").context("outermost");
|
||||
assert!(is_process_timeout(&wrapped));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_with_timeout_reports_nonzero_exit_with_stderr_tail() {
|
||||
let err = run_with_timeout(
|
||||
Path::new("sh"),
|
||||
&os(&["-c", "echo first-line >&2; echo boom-at-the-end >&2; exit 3"]),
|
||||
None,
|
||||
Duration::from_secs(30),
|
||||
)
|
||||
.unwrap_err();
|
||||
let msg = format!("{err:#}");
|
||||
assert!(msg.contains("exited with"), "{msg}");
|
||||
assert!(msg.contains("boom-at-the-end"), "{msg}");
|
||||
assert!(!is_process_timeout(&err));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_with_timeout_round_trips_stdin() {
|
||||
let out = run_with_timeout(
|
||||
Path::new("cat"),
|
||||
&[],
|
||||
Some("prompt text"),
|
||||
Duration::from_secs(30),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(out.stdout, "prompt text");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn last_chars_keeps_the_tail() {
|
||||
assert_eq!(last_chars("abc", 5), "abc");
|
||||
assert_eq!(last_chars("abcdef", 3), "…def");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn run_with_timeout_kills_grandchildren_on_timeout() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pid_file = dir.path().join("grandchild.pid");
|
||||
let script = format!("{}sleep 30", test_support::spawn_grandchild_snippet(&pid_file));
|
||||
let started = Instant::now();
|
||||
let err = run_with_timeout(Path::new("sh"), &os(&["-c", &script]), None, Duration::from_secs(1))
|
||||
.unwrap_err();
|
||||
assert!(is_process_timeout(&err), "{err:#}");
|
||||
assert!(started.elapsed() < Duration::from_secs(5));
|
||||
test_support::assert_grandchild_gone(&pid_file);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn run_with_timeout_does_not_wait_out_budget_for_pipe_holding_grandchild() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pid_file = dir.path().join("grandchild.pid");
|
||||
let script = format!("{}echo done", test_support::spawn_grandchild_snippet(&pid_file));
|
||||
let started = Instant::now();
|
||||
let out = run_with_timeout(Path::new("sh"), &os(&["-c", &script]), None, Duration::from_secs(60))
|
||||
.unwrap();
|
||||
assert_eq!(out.stdout, "done\n");
|
||||
assert!(started.elapsed() < Duration::from_secs(10), "{:?}", started.elapsed());
|
||||
test_support::assert_grandchild_gone(&pid_file);
|
||||
}
|
||||
}
|
||||
844
crates/archivr-core/src/subtitles.rs
Normal file
844
crates/archivr-core/src/subtitles.rs
Normal file
|
|
@ -0,0 +1,844 @@
|
|||
//! Subtitle artifacts: archiving staged yt-dlp subtitle files, registering
|
||||
//! them as `subtitle` artifacts, fetching them on demand for existing entries,
|
||||
//! ranking tracks, and reducing VTT/SRT to a plain transcript for summaries.
|
||||
|
||||
use anyhow::{anyhow, Context, Result};
|
||||
use regex::Regex;
|
||||
use rusqlite::{Connection, Transaction, TransactionBehavior};
|
||||
use std::{
|
||||
fs,
|
||||
path::{Path, PathBuf},
|
||||
sync::LazyLock,
|
||||
};
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::archive::ArchivePaths;
|
||||
use crate::capture;
|
||||
use crate::database::{self, BlobRecord, NewArtifact};
|
||||
use crate::downloader::store;
|
||||
use crate::downloader::ytdlp::{self, language_base, StagedSubtitle, SubtitleKind};
|
||||
|
||||
pub const SUBTITLE_ARTIFACT_ROLE: &str = "subtitle";
|
||||
pub const SUBTITLE_ORIGIN_CAPTURE: &str = "capture";
|
||||
pub const SUBTITLE_ORIGIN_SUMMARY_FETCH: &str = "summary_fetch";
|
||||
/// Origin of a track produced by local transcription (`kind: "transcribed"`).
|
||||
pub const SUBTITLE_ORIGIN_TRANSCRIPTION: &str = "transcription";
|
||||
|
||||
/// Result of [`fetch_subtitles_for_entry`].
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||
pub struct SubtitleFetchOutcome {
|
||||
/// Artifact rows inserted by this call.
|
||||
pub added: usize,
|
||||
/// The video's original language, from a successful metadata probe or
|
||||
/// else from existing subtitle artifacts' metadata.
|
||||
pub original_language: Option<String>,
|
||||
}
|
||||
|
||||
/// Subtitle file formats archivr keeps.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum SubtitleFormat {
|
||||
Vtt,
|
||||
Srt,
|
||||
}
|
||||
|
||||
impl SubtitleFormat {
|
||||
/// Detects the format from a file extension (with or without the dot),
|
||||
/// falling back to the MIME type. Case-insensitive.
|
||||
pub fn detect(extension: &str, mime: &str) -> Option<Self> {
|
||||
let ext = extension.trim_start_matches('.').to_ascii_lowercase();
|
||||
match ext.as_str() {
|
||||
"vtt" => return Some(SubtitleFormat::Vtt),
|
||||
"srt" => return Some(SubtitleFormat::Srt),
|
||||
_ => {}
|
||||
}
|
||||
let mime = mime.split(';').next().unwrap_or("").trim().to_ascii_lowercase();
|
||||
match mime.as_str() {
|
||||
"text/vtt" => Some(SubtitleFormat::Vtt),
|
||||
"application/x-subrip" => Some(SubtitleFormat::Srt),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn mime(self) -> &'static str {
|
||||
match self {
|
||||
SubtitleFormat::Vtt => "text/vtt",
|
||||
SubtitleFormat::Srt => "application/x-subrip",
|
||||
}
|
||||
}
|
||||
|
||||
pub fn extension(self) -> &'static str {
|
||||
match self {
|
||||
SubtitleFormat::Vtt => "vtt",
|
||||
SubtitleFormat::Srt => "srt",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A subtitle file already moved into the content-addressed `raw/` store.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct ArchivedSubtitle {
|
||||
/// Store-relative path, e.g. `raw/a/b/<hash>.vtt`.
|
||||
pub raw_relpath: PathBuf,
|
||||
pub language: String,
|
||||
pub kind: SubtitleKind,
|
||||
pub format: SubtitleFormat,
|
||||
pub original_language: Option<String>,
|
||||
}
|
||||
|
||||
/// Track description parsed from a `subtitle` artifact's `metadata_json`.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct SubtitleTrackMeta {
|
||||
pub language: String,
|
||||
pub kind: SubtitleKind,
|
||||
pub original_language: Option<String>,
|
||||
}
|
||||
|
||||
/// Moves staged subtitle files into `raw/`. Files that fail to archive or have
|
||||
/// an unsupported format are logged and skipped — subtitles never fail a capture.
|
||||
pub fn archive_staged_subtitles(
|
||||
store_path: &Path,
|
||||
staged: Vec<StagedSubtitle>,
|
||||
) -> Vec<ArchivedSubtitle> {
|
||||
let mut archived = Vec::with_capacity(staged.len());
|
||||
for sub in staged {
|
||||
let Some(format) = SubtitleFormat::detect(&sub.format, "") else {
|
||||
eprintln!(
|
||||
"warn: archive subtitle {}: unsupported format {}",
|
||||
sub.path.display(),
|
||||
sub.format
|
||||
);
|
||||
continue;
|
||||
};
|
||||
match store::archive_staged_file(&sub.path, store_path) {
|
||||
Ok(raw_relpath) => archived.push(ArchivedSubtitle {
|
||||
raw_relpath,
|
||||
language: sub.language,
|
||||
kind: sub.kind,
|
||||
format,
|
||||
original_language: sub.original_language,
|
||||
}),
|
||||
Err(e) => eprintln!("warn: archive subtitle {}: {e:#}", sub.path.display()),
|
||||
}
|
||||
}
|
||||
archived
|
||||
}
|
||||
|
||||
/// Registers archived subtitles as `subtitle` artifacts of `entry_id`.
|
||||
///
|
||||
/// Runs in one `BEGIN IMMEDIATE` transaction so concurrent registrations of
|
||||
/// the same content serialize; an `(entry, subtitle, blob)` that already
|
||||
/// exists is skipped. Returns the number of artifact rows inserted.
|
||||
pub fn register_subtitle_artifacts(
|
||||
conn: &Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
subtitles: &[ArchivedSubtitle],
|
||||
origin: &str,
|
||||
) -> Result<usize> {
|
||||
let rows: Vec<(&ArchivedSubtitle, serde_json::Value)> = subtitles
|
||||
.iter()
|
||||
.map(|sub| {
|
||||
let metadata = serde_json::json!({
|
||||
"language": sub.language,
|
||||
"kind": sub.kind.as_str(),
|
||||
"format": sub.format.extension(),
|
||||
"original_language": sub.original_language,
|
||||
"origin": origin,
|
||||
});
|
||||
(sub, metadata)
|
||||
})
|
||||
.collect();
|
||||
insert_subtitle_rows(conn, store_path, entry_id, &rows)
|
||||
}
|
||||
|
||||
/// Registers a locally transcribed track (origin `transcription`), recording
|
||||
/// the engine kind and model. A model given as a filesystem path is stored as
|
||||
/// its file name only, so no host path is persisted. Same transaction and
|
||||
/// dedup rules as [`register_subtitle_artifacts`].
|
||||
pub fn register_transcript_artifact(
|
||||
conn: &Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
sub: &ArchivedSubtitle,
|
||||
engine: &str,
|
||||
model: &str,
|
||||
) -> Result<usize> {
|
||||
let metadata = serde_json::json!({
|
||||
"language": sub.language,
|
||||
"kind": sub.kind.as_str(),
|
||||
"format": sub.format.extension(),
|
||||
"original_language": sub.original_language,
|
||||
"origin": SUBTITLE_ORIGIN_TRANSCRIPTION,
|
||||
"engine": engine,
|
||||
"model": sanitize_model_name(model),
|
||||
});
|
||||
insert_subtitle_rows(conn, store_path, entry_id, &[(sub, metadata)])
|
||||
}
|
||||
|
||||
/// Reduces a model that is a filesystem path (contains `\`, is absolute, or
|
||||
/// exists) to its file name. Hugging Face ids such as
|
||||
/// `nvidia/parakeet-tdt-0.6b-v3` are kept as-is.
|
||||
pub(crate) fn sanitize_model_name(model: &str) -> String {
|
||||
let path = Path::new(model);
|
||||
if model.contains('\\') || path.is_absolute() || path.exists() {
|
||||
let name = model.rsplit(['/', '\\']).next().unwrap_or(model);
|
||||
if !name.is_empty() {
|
||||
return name.to_string();
|
||||
}
|
||||
}
|
||||
model.to_string()
|
||||
}
|
||||
|
||||
/// Inserts one `subtitle` artifact per row with the given metadata, in one
|
||||
/// `BEGIN IMMEDIATE` transaction; rows whose blob is already a subtitle of
|
||||
/// the entry, or whose file can't be stat'ed, are skipped. Refreshes the
|
||||
/// entry's cached bytes and returns the number of rows inserted.
|
||||
fn insert_subtitle_rows(
|
||||
conn: &Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
rows: &[(&ArchivedSubtitle, serde_json::Value)],
|
||||
) -> Result<usize> {
|
||||
if rows.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
// No transaction is open on `conn` at any call site, so new_unchecked is safe.
|
||||
let tx = Transaction::new_unchecked(conn, TransactionBehavior::Immediate)?;
|
||||
let mut inserted = 0;
|
||||
for (sub, metadata) in rows {
|
||||
let relpath = sub.raw_relpath.to_string_lossy().replace('\\', "/");
|
||||
let sha256 = sub
|
||||
.raw_relpath
|
||||
.file_stem()
|
||||
.and_then(|s| s.to_str())
|
||||
.with_context(|| format!("subtitle path has no hash stem: {relpath}"))?
|
||||
.to_string();
|
||||
let byte_size = match fs::metadata(store_path.join(&sub.raw_relpath)) {
|
||||
Ok(meta) => meta.len() as i64,
|
||||
Err(e) => {
|
||||
eprintln!("warn: skipping archived subtitle {relpath}: {e:#}");
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let blob_id = database::upsert_blob(
|
||||
&tx,
|
||||
&BlobRecord {
|
||||
sha256,
|
||||
byte_size,
|
||||
mime_type: Some(sub.format.mime().to_string()),
|
||||
extension: Some(sub.format.extension().to_string()),
|
||||
raw_relpath: relpath.clone(),
|
||||
},
|
||||
)?;
|
||||
if database::entry_has_artifact_blob(&tx, entry_id, SUBTITLE_ARTIFACT_ROLE, blob_id)? {
|
||||
continue;
|
||||
}
|
||||
database::add_entry_artifact(
|
||||
&tx,
|
||||
&NewArtifact {
|
||||
entry_id,
|
||||
artifact_role: SUBTITLE_ARTIFACT_ROLE.to_string(),
|
||||
storage_area: "raw".to_string(),
|
||||
relpath,
|
||||
blob_id: Some(blob_id),
|
||||
logical_path: None,
|
||||
metadata_json: Some(metadata.to_string()),
|
||||
},
|
||||
)?;
|
||||
inserted += 1;
|
||||
}
|
||||
tx.commit()?;
|
||||
database::refresh_entry_cached_bytes(conn, entry_id)?;
|
||||
Ok(inserted)
|
||||
}
|
||||
|
||||
/// Number of the entry's `subtitle` artifacts that reduce to a non-empty
|
||||
/// transcript. Unreadable or unsupported files do not count.
|
||||
pub(crate) fn usable_subtitle_count(
|
||||
conn: &Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
) -> Result<usize> {
|
||||
let artifacts = database::list_entry_artifacts_by_role(conn, entry_id, SUBTITLE_ARTIFACT_ROLE)?;
|
||||
Ok(artifacts
|
||||
.iter()
|
||||
.filter(|a| {
|
||||
let ext = Path::new(&a.relpath)
|
||||
.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.unwrap_or("");
|
||||
SubtitleFormat::detect(ext, a.mime_type.as_deref().unwrap_or("")).is_some()
|
||||
&& fs::read_to_string(store_path.join(&a.relpath))
|
||||
.map(|raw| !subtitle_to_transcript(&raw).is_empty())
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.count())
|
||||
}
|
||||
|
||||
/// `original_language` of the entry's first `subtitle` artifact (id order)
|
||||
/// whose metadata records one.
|
||||
fn existing_original_language(conn: &Connection, entry_id: i64) -> Result<Option<String>> {
|
||||
let artifacts = database::list_entry_artifacts_by_role(conn, entry_id, SUBTITLE_ARTIFACT_ROLE)?;
|
||||
Ok(artifacts
|
||||
.iter()
|
||||
.find_map(|a| parse_subtitle_metadata(a.metadata_json.as_deref()).original_language))
|
||||
}
|
||||
|
||||
/// Downloads subtitles for an existing YouTube video entry from its original
|
||||
/// URL and registers them. Returns the number of rows added by this call plus
|
||||
/// the video's original language, taken from the metadata probe when it
|
||||
/// succeeds and otherwise from existing subtitle artifacts.
|
||||
///
|
||||
/// Returns `added: 0` (no yt-dlp call) for non-YouTube-video entries or a
|
||||
/// missing / non-http(s) canonical URL, and without fetching when the entry
|
||||
/// already has a usable subtitle (concurrency re-check). An unreachable video
|
||||
/// or any yt-dlp failure is logged and counts as zero subtitles; only DB/IO
|
||||
/// errors propagate.
|
||||
pub fn fetch_subtitles_for_entry(
|
||||
paths: &ArchivePaths,
|
||||
entry_uid: &str,
|
||||
cookie_rules: &[database::CookieRule],
|
||||
) -> Result<SubtitleFetchOutcome> {
|
||||
let conn = database::open_or_initialize(&paths.archive_path)?;
|
||||
let info = database::entry_source_info(&conn, entry_uid)?
|
||||
.ok_or_else(|| anyhow!("entry not found: {entry_uid}"))?;
|
||||
let existing_language = existing_original_language(&conn, info.entry_id)?;
|
||||
let nothing_added = || SubtitleFetchOutcome {
|
||||
added: 0,
|
||||
original_language: existing_language.clone(),
|
||||
};
|
||||
if info.source_kind != "youtube" || info.entity_kind != "video" {
|
||||
return Ok(nothing_added());
|
||||
}
|
||||
let Some(url) = info
|
||||
.canonical_url
|
||||
.filter(|u| u.starts_with("https://") || u.starts_with("http://"))
|
||||
else {
|
||||
return Ok(nothing_added());
|
||||
};
|
||||
|
||||
let store_path = &paths.store_path;
|
||||
if usable_subtitle_count(&conn, store_path, info.entry_id)? > 0 {
|
||||
return Ok(nothing_added());
|
||||
}
|
||||
|
||||
let cookies = capture::resolve_cookies_for_url(cookie_rules, &url);
|
||||
let timeout = crate::summarizer::summary_cli_timeout();
|
||||
let Some(metadata) = ytdlp::fetch_metadata_with_timeout(&url, &cookies, Some(timeout)) else {
|
||||
eprintln!("warn: subtitle fetch for {entry_uid}: video unreachable ({url})");
|
||||
return Ok(nothing_added());
|
||||
};
|
||||
// Before planning: a video without captions is exactly the case that
|
||||
// needs its language for a transcription fallback.
|
||||
let original_language = serde_json::from_str::<serde_json::Value>(&metadata)
|
||||
.ok()
|
||||
.and_then(|v| ytdlp::original_language_from_metadata(&v))
|
||||
.or_else(|| existing_language.clone());
|
||||
let Some(request) = ytdlp::plan_subtitle_request(Some(&metadata)) else {
|
||||
eprintln!("info: subtitle fetch for {entry_uid}: no subtitle tracks available ({url})");
|
||||
return Ok(SubtitleFetchOutcome {
|
||||
added: 0,
|
||||
original_language,
|
||||
});
|
||||
};
|
||||
|
||||
let stage_key = format!("subs-{}", Uuid::new_v4().simple());
|
||||
let stage_dir = store_path.join("temp").join(&stage_key);
|
||||
let staged = match ytdlp::download_subtitles(&url, store_path, &stage_key, &request, &cookies, timeout) {
|
||||
Ok(staged) => staged,
|
||||
Err(e) => {
|
||||
eprintln!("warn: subtitle fetch for {entry_uid} failed: {e:#}");
|
||||
let _ = fs::remove_dir_all(&stage_dir);
|
||||
return Ok(SubtitleFetchOutcome {
|
||||
added: 0,
|
||||
original_language,
|
||||
});
|
||||
}
|
||||
};
|
||||
let archived = archive_staged_subtitles(store_path, staged);
|
||||
let _ = fs::remove_dir_all(&stage_dir);
|
||||
|
||||
let added = register_subtitle_artifacts(
|
||||
&conn,
|
||||
store_path,
|
||||
info.entry_id,
|
||||
&archived,
|
||||
SUBTITLE_ORIGIN_SUMMARY_FETCH,
|
||||
)?;
|
||||
eprintln!("info: subtitle fetch for {entry_uid}: registered {added} subtitle artifact(s)");
|
||||
Ok(SubtitleFetchOutcome {
|
||||
added,
|
||||
original_language,
|
||||
})
|
||||
}
|
||||
|
||||
/// Any `<...>` markup: `<c>`, `<c.colorE5E5E5>`, `<00:00:01.000>`, `<v Speaker>`, `<i>`.
|
||||
static TAG_RE: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"<[^>]*>").expect("valid tag regex"));
|
||||
|
||||
/// SRT ASS override blocks such as `{\an8}`.
|
||||
static ASS_OVERRIDE_RE: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"\{\\[^}]*\}").expect("valid ASS override regex"));
|
||||
|
||||
/// Strips markup from one cue text line and normalizes whitespace.
|
||||
fn clean_cue_line(line: &str) -> String {
|
||||
let without_tags = TAG_RE.replace_all(line, "");
|
||||
let without_ass = ASS_OVERRIDE_RE.replace_all(&without_tags, "");
|
||||
// `&` last so `&lt;` decodes to `<`, not `<`.
|
||||
let decoded = without_ass
|
||||
.replace("<", "<")
|
||||
.replace(">", ">")
|
||||
.replace(""", "\"")
|
||||
.replace("'", "'")
|
||||
.replace(" ", " ")
|
||||
.replace("&", "&");
|
||||
decoded.split_whitespace().collect::<Vec<_>>().join(" ")
|
||||
}
|
||||
|
||||
/// Appends `line` unless it repeats one of the last two lines; a line that
|
||||
/// extends the previous one (rolling auto-captions) replaces it.
|
||||
fn push_deduped(out: &mut Vec<String>, line: String) {
|
||||
let recent = &out[out.len().saturating_sub(2)..];
|
||||
if recent.iter().any(|l| *l == line) {
|
||||
return;
|
||||
}
|
||||
if let Some(last) = out.last_mut() {
|
||||
if line.len() > last.len() && line.starts_with(last.as_str()) {
|
||||
*last = line;
|
||||
return;
|
||||
}
|
||||
}
|
||||
out.push(line);
|
||||
}
|
||||
|
||||
/// Text lines of one cue block (everything after its timing line).
|
||||
fn reduce_block(block: &[&str], out: &mut Vec<String>) {
|
||||
let Some(timing) = block.iter().position(|l| l.contains("-->")) else {
|
||||
return; // WEBVTT header, NOTE, STYLE, REGION, bare index
|
||||
};
|
||||
for line in &block[timing + 1..] {
|
||||
let cleaned = clean_cue_line(line);
|
||||
if !cleaned.is_empty() {
|
||||
push_deduped(out, cleaned);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Reduces a VTT or SRT document to plain transcript text, one line per
|
||||
/// caption line, with markup, timings and rolling-caption repeats removed.
|
||||
///
|
||||
/// Blocks split only on truly empty lines: YouTube auto-caption cues contain
|
||||
/// lines holding a single space, which belong to the cue.
|
||||
pub fn subtitle_to_transcript(raw: &str) -> String {
|
||||
let text = raw
|
||||
.strip_prefix('\u{feff}')
|
||||
.unwrap_or(raw)
|
||||
.replace("\r\n", "\n")
|
||||
.replace('\r', "\n");
|
||||
let mut out: Vec<String> = Vec::new();
|
||||
let mut block: Vec<&str> = Vec::new();
|
||||
for line in text.split('\n') {
|
||||
if line.is_empty() {
|
||||
reduce_block(&block, &mut out);
|
||||
block.clear();
|
||||
} else {
|
||||
block.push(line);
|
||||
}
|
||||
}
|
||||
reduce_block(&block, &mut out);
|
||||
out.join("\n")
|
||||
}
|
||||
|
||||
/// Parses a `subtitle` artifact's `metadata_json`. Missing or invalid metadata
|
||||
/// yields language `""` and `Unknown` kind.
|
||||
pub fn parse_subtitle_metadata(metadata_json: Option<&str>) -> SubtitleTrackMeta {
|
||||
let value: serde_json::Value = metadata_json
|
||||
.and_then(|json| serde_json::from_str(json).ok())
|
||||
.unwrap_or(serde_json::Value::Null);
|
||||
let text = |key: &str| {
|
||||
value
|
||||
.get(key)
|
||||
.and_then(|v| v.as_str())
|
||||
.map(str::trim)
|
||||
.filter(|s| !s.is_empty())
|
||||
.map(str::to_string)
|
||||
};
|
||||
SubtitleTrackMeta {
|
||||
language: text("language").unwrap_or_default(),
|
||||
kind: text("kind")
|
||||
.map(|k| SubtitleKind::parse(&k))
|
||||
.unwrap_or(SubtitleKind::Unknown),
|
||||
original_language: text("original_language"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Preference rank of a subtitle track for summaries; lower is better.
|
||||
///
|
||||
/// 0 manual English, 1 manual original-language, 2 other manual,
|
||||
/// 3 transcribed (any language), 4 auto/unknown original-language,
|
||||
/// 5 auto/unknown English, 6 anything else.
|
||||
pub fn subtitle_track_rank(meta: &SubtitleTrackMeta) -> u8 {
|
||||
let base = language_base(&meta.language);
|
||||
let is_en = base == "en";
|
||||
let is_orig = meta.language.to_ascii_lowercase().ends_with("-orig")
|
||||
|| (!base.is_empty()
|
||||
&& meta
|
||||
.original_language
|
||||
.as_deref()
|
||||
.is_some_and(|orig| language_base(orig) == base));
|
||||
match meta.kind {
|
||||
SubtitleKind::Manual if is_en => 0,
|
||||
SubtitleKind::Manual if is_orig => 1,
|
||||
SubtitleKind::Manual => 2,
|
||||
SubtitleKind::Transcribed => 3,
|
||||
_ if is_orig => 4,
|
||||
_ if is_en => 5,
|
||||
_ => 6,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn meta(language: &str, kind: SubtitleKind, original: Option<&str>) -> SubtitleTrackMeta {
|
||||
SubtitleTrackMeta {
|
||||
language: language.to_string(),
|
||||
kind,
|
||||
original_language: original.map(str::to_string),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vtt_reduction_strips_header_timestamps_settings_and_tags() {
|
||||
let vtt = "WEBVTT\nKind: captions\nLanguage: en\n\nNOTE a comment\nspanning lines\n\nSTYLE\n::cue { color: red }\n\ncue-1\n00:00:01.000 --> 00:00:03.000 align:start position:0%\n<v Speaker>Hello <i>there</i></v>\n\n00:00:03.000 --> 00:00:05.000\n<c.colorE5E5E5>General</c> <00:00:03.500><c>Kenobi</c>\n";
|
||||
assert_eq!(subtitle_to_transcript(vtt), "Hello there\nGeneral Kenobi");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vtt_reduction_collapses_rolling_auto_captions() {
|
||||
// Real-shaped YouTube auto-caption VTT: each cue repeats the previous
|
||||
// line, 10 ms "freeze" cues duplicate it, and lines holding a single
|
||||
// space sit inside cues (they must not split blocks).
|
||||
let vtt = "WEBVTT\nKind: captions\nLanguage: en\n\n\
|
||||
00:00:00.000 --> 00:00:02.030 align:start position:0%\n \nhello<00:00:00.320><c> world</c><00:00:00.640><c> this</c>\n\n\
|
||||
00:00:02.030 --> 00:00:02.040 align:start position:0%\nhello world this\n \n\n\
|
||||
00:00:02.040 --> 00:00:04.110 align:start position:0%\nhello world this\nis<00:00:02.360><c> a</c><00:00:02.600><c> test</c>\n\n\
|
||||
00:00:04.110 --> 00:00:04.120 align:start position:0%\nis a test\n \n\n\
|
||||
00:00:04.120 --> 00:00:06.000 align:start position:0%\nis a test\nof<00:00:04.500><c> captions</c>\n\n\
|
||||
00:00:06.000 --> 00:00:06.010 align:start position:0%\nof captions\n \n";
|
||||
assert_eq!(
|
||||
subtitle_to_transcript(vtt),
|
||||
"hello world this\nis a test\nof captions"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vtt_reduction_extends_growing_lines() {
|
||||
let vtt = "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nhello\n\n00:00:01.000 --> 00:00:02.000\nhello world\n";
|
||||
assert_eq!(subtitle_to_transcript(vtt), "hello world");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn srt_reduction_strips_indices_italics_and_ass_overrides() {
|
||||
let srt = "1\n00:00:01,000 --> 00:00:02,000\n<i>Hello</i> there\n\n2\n00:00:02,500 --> 00:00:04,000\n{\\an8}Second line\n<b>continues</b> here\n\n3\n00:00:04,000 --> 00:00:05,000\n42\n";
|
||||
assert_eq!(
|
||||
subtitle_to_transcript(srt),
|
||||
"Hello there\nSecond line\ncontinues here\n42"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reduction_handles_bom_crlf_and_entities() {
|
||||
let vtt = "\u{feff}WEBVTT\r\n\r\n00:00:01.000 --> 00:00:02.000\r\nTom & Jerry <3 "cheese" it's &lt;\r\n\r\n00:00:02.000 --> 00:00:03.000\rold mac line\r";
|
||||
assert_eq!(
|
||||
subtitle_to_transcript(vtt),
|
||||
"Tom & Jerry <3 \"cheese\" it's <\nold mac line"
|
||||
);
|
||||
assert_eq!(subtitle_to_transcript(""), "");
|
||||
assert_eq!(subtitle_to_transcript("WEBVTT\n\n"), "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn track_rank_prefers_manual_english_then_manual_original_then_auto_original() {
|
||||
let de = Some("de");
|
||||
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Manual, de)), 0);
|
||||
assert_eq!(subtitle_track_rank(&meta("en-GB", SubtitleKind::Manual, de)), 0);
|
||||
assert_eq!(subtitle_track_rank(&meta("de", SubtitleKind::Manual, de)), 1);
|
||||
assert_eq!(subtitle_track_rank(&meta("fr", SubtitleKind::Manual, de)), 2);
|
||||
assert_eq!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Auto, de)), 4);
|
||||
assert_eq!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Unknown, None)), 4);
|
||||
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Auto, de)), 5);
|
||||
assert_eq!(subtitle_track_rank(&meta("fr", SubtitleKind::Auto, de)), 6);
|
||||
assert_eq!(subtitle_track_rank(&meta("", SubtitleKind::Unknown, None)), 6);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn track_rank_places_transcribed_below_manual_above_auto() {
|
||||
let de = Some("de");
|
||||
let transcribed = subtitle_track_rank(&meta("fr", SubtitleKind::Transcribed, de));
|
||||
assert_eq!(transcribed, 3);
|
||||
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Transcribed, None)), 3);
|
||||
assert!(subtitle_track_rank(&meta("fr", SubtitleKind::Manual, de)) < transcribed);
|
||||
assert!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Auto, de)) > transcribed);
|
||||
assert!(subtitle_track_rank(&meta("en", SubtitleKind::Auto, de)) > transcribed);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_subtitle_metadata_defaults_and_round_trip() {
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(None),
|
||||
meta("", SubtitleKind::Unknown, None)
|
||||
);
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(Some("not json")),
|
||||
meta("", SubtitleKind::Unknown, None)
|
||||
);
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(Some(
|
||||
r#"{"language":"de-orig","kind":"auto","format":"vtt","original_language":"de","origin":"capture"}"#
|
||||
)),
|
||||
meta("de-orig", SubtitleKind::Auto, Some("de"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subtitle_format_detects_by_extension_or_mime() {
|
||||
assert_eq!(SubtitleFormat::detect("vtt", ""), Some(SubtitleFormat::Vtt));
|
||||
assert_eq!(SubtitleFormat::detect(".SRT", ""), Some(SubtitleFormat::Srt));
|
||||
assert_eq!(
|
||||
SubtitleFormat::detect("", "text/vtt; charset=utf-8"),
|
||||
Some(SubtitleFormat::Vtt)
|
||||
);
|
||||
assert_eq!(
|
||||
SubtitleFormat::detect("txt", "application/x-subrip"),
|
||||
Some(SubtitleFormat::Srt)
|
||||
);
|
||||
assert_eq!(SubtitleFormat::detect("ttml", "application/ttml+xml"), None);
|
||||
}
|
||||
|
||||
fn archive_fixture(
|
||||
source_kind: &str,
|
||||
entity_kind: &str,
|
||||
canonical_url: Option<&str>,
|
||||
) -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let paths = crate::archive::initialize_archive(
|
||||
temp.path(),
|
||||
&temp.path().join("store"),
|
||||
"Test archive",
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
let user_id = database::ensure_default_user(&conn).unwrap();
|
||||
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
|
||||
let source_id = database::upsert_source_identity(
|
||||
&conn,
|
||||
source_kind,
|
||||
entity_kind,
|
||||
Some("fixture-1"),
|
||||
canonical_url,
|
||||
canonical_url.unwrap_or("fixture:1"),
|
||||
)
|
||||
.unwrap();
|
||||
let entry = database::create_archived_entry(
|
||||
&conn,
|
||||
&database::NewEntry {
|
||||
source_identity_id: source_id,
|
||||
archive_run_id: run.id,
|
||||
parent_entry_id: None,
|
||||
root_entry_id: None,
|
||||
created_by_user_id: user_id,
|
||||
owned_by_user_id: user_id,
|
||||
source_kind: source_kind.to_string(),
|
||||
entity_kind: entity_kind.to_string(),
|
||||
title: None,
|
||||
visibility: "private".to_string(),
|
||||
representation_kind: entity_kind.to_string(),
|
||||
source_metadata_json: "{}".to_string(),
|
||||
display_metadata_json: None,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
(temp, paths, entry)
|
||||
}
|
||||
|
||||
fn stage_vtt(store_path: &Path, name: &str, body: &str) -> StagedSubtitle {
|
||||
let dir = store_path.join("temp").join("stage");
|
||||
fs::create_dir_all(&dir).unwrap();
|
||||
let path = dir.join(name);
|
||||
fs::write(&path, body).unwrap();
|
||||
StagedSubtitle {
|
||||
path,
|
||||
language: "de-orig".to_string(),
|
||||
kind: SubtitleKind::Auto,
|
||||
format: "vtt".to_string(),
|
||||
original_language: Some("de".to_string()),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn register_subtitle_artifacts_dedups_same_blob() {
|
||||
let (_temp, paths, entry) =
|
||||
archive_fixture("youtube", "video", Some("https://www.youtube.com/watch?v=x"));
|
||||
let store_path = &paths.store_path;
|
||||
let body = "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHallo Welt\n";
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
|
||||
let first = archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "a.de-orig.vtt", body)]);
|
||||
assert_eq!(first.len(), 1);
|
||||
assert!(store_path.join(&first[0].raw_relpath).is_file());
|
||||
assert_eq!(
|
||||
register_subtitle_artifacts(&conn, store_path, entry.id, &first, SUBTITLE_ORIGIN_CAPTURE)
|
||||
.unwrap(),
|
||||
1
|
||||
);
|
||||
|
||||
// Same bytes fetched again: raw move dedupes, registration skips.
|
||||
let second = archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "b.de-orig.vtt", body)]);
|
||||
assert_eq!(second[0].raw_relpath, first[0].raw_relpath);
|
||||
assert_eq!(
|
||||
register_subtitle_artifacts(
|
||||
&conn,
|
||||
store_path,
|
||||
entry.id,
|
||||
&second,
|
||||
SUBTITLE_ORIGIN_SUMMARY_FETCH
|
||||
)
|
||||
.unwrap(),
|
||||
0
|
||||
);
|
||||
|
||||
let rows =
|
||||
database::list_entry_artifacts_by_role(&conn, entry.id, SUBTITLE_ARTIFACT_ROLE).unwrap();
|
||||
assert_eq!(rows.len(), 1);
|
||||
assert_eq!(rows[0].mime_type.as_deref(), Some("text/vtt"));
|
||||
assert_eq!(rows[0].relpath, first[0].raw_relpath.to_string_lossy());
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(rows[0].metadata_json.as_deref()),
|
||||
meta("de-orig", SubtitleKind::Auto, Some("de"))
|
||||
);
|
||||
let stored: serde_json::Value =
|
||||
serde_json::from_str(rows[0].metadata_json.as_deref().unwrap()).unwrap();
|
||||
assert_eq!(stored["format"], "vtt");
|
||||
assert_eq!(stored["origin"], SUBTITLE_ORIGIN_CAPTURE);
|
||||
|
||||
assert_eq!(usable_subtitle_count(&conn, store_path, entry.id).unwrap(), 1);
|
||||
assert_eq!(
|
||||
register_subtitle_artifacts(&conn, store_path, entry.id, &[], SUBTITLE_ORIGIN_CAPTURE)
|
||||
.unwrap(),
|
||||
0
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fetch_subtitles_for_entry_skips_non_youtube_and_non_http_entries() {
|
||||
// Each of these returns before any yt-dlp process could be spawned.
|
||||
let (_t1, web_paths, web) = archive_fixture("web", "page", Some("https://example.com/"));
|
||||
assert_eq!(
|
||||
fetch_subtitles_for_entry(&web_paths, &web.entry_uid, &[]).unwrap(),
|
||||
SubtitleFetchOutcome::default()
|
||||
);
|
||||
|
||||
let (_t2, offline_paths, offline) =
|
||||
archive_fixture("youtube", "video", Some("youtube-test:offline"));
|
||||
assert_eq!(
|
||||
fetch_subtitles_for_entry(&offline_paths, &offline.entry_uid, &[]).unwrap(),
|
||||
SubtitleFetchOutcome::default()
|
||||
);
|
||||
|
||||
let (_t3, no_url_paths, no_url) = archive_fixture("youtube", "video", None);
|
||||
assert_eq!(
|
||||
fetch_subtitles_for_entry(&no_url_paths, &no_url.entry_uid, &[]).unwrap(),
|
||||
SubtitleFetchOutcome::default()
|
||||
);
|
||||
|
||||
assert!(fetch_subtitles_for_entry(&web_paths, "entry_missing", &[]).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fetch_outcome_reports_original_language_from_existing_artifacts() {
|
||||
let (_temp, paths, entry) =
|
||||
archive_fixture("youtube", "video", Some("youtube-test:offline"));
|
||||
let store_path = &paths.store_path;
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
// An unusable (empty) track still carries the original language.
|
||||
let archived =
|
||||
archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "e.de-orig.vtt", "WEBVTT\n")]);
|
||||
register_subtitle_artifacts(&conn, store_path, entry.id, &archived, SUBTITLE_ORIGIN_CAPTURE)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
fetch_subtitles_for_entry(&paths, &entry.entry_uid, &[]).unwrap(),
|
||||
SubtitleFetchOutcome {
|
||||
added: 0,
|
||||
original_language: Some("de".to_string()),
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn register_transcript_artifact_writes_engine_metadata_and_dedups() {
|
||||
let (_temp, paths, entry) =
|
||||
archive_fixture("youtube", "video", Some("https://www.youtube.com/watch?v=x"));
|
||||
let store_path = &paths.store_path;
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
let body = "WEBVTT\n\n00:00:00.000 --> 00:00:02.000\nhello world\n";
|
||||
let staged = |name: &str| {
|
||||
let mut s = stage_vtt(store_path, name, body);
|
||||
s.language = "en".to_string();
|
||||
s.kind = SubtitleKind::Transcribed;
|
||||
s.original_language = None;
|
||||
s
|
||||
};
|
||||
|
||||
let first = archive_staged_subtitles(store_path, vec![staged("t1.vtt")]);
|
||||
assert_eq!(first.len(), 1);
|
||||
assert_eq!(
|
||||
register_transcript_artifact(
|
||||
&conn,
|
||||
store_path,
|
||||
entry.id,
|
||||
&first[0],
|
||||
"whisper",
|
||||
"/models/ggml-tiny.bin"
|
||||
)
|
||||
.unwrap(),
|
||||
1
|
||||
);
|
||||
let second = archive_staged_subtitles(store_path, vec![staged("t2.vtt")]);
|
||||
assert_eq!(
|
||||
register_transcript_artifact(
|
||||
&conn,
|
||||
store_path,
|
||||
entry.id,
|
||||
&second[0],
|
||||
"whisper",
|
||||
"/models/ggml-tiny.bin"
|
||||
)
|
||||
.unwrap(),
|
||||
0
|
||||
);
|
||||
|
||||
let rows =
|
||||
database::list_entry_artifacts_by_role(&conn, entry.id, SUBTITLE_ARTIFACT_ROLE).unwrap();
|
||||
assert_eq!(rows.len(), 1);
|
||||
assert_eq!(rows[0].mime_type.as_deref(), Some("text/vtt"));
|
||||
let stored: serde_json::Value =
|
||||
serde_json::from_str(rows[0].metadata_json.as_deref().unwrap()).unwrap();
|
||||
assert_eq!(stored["kind"], "transcribed");
|
||||
assert_eq!(stored["origin"], SUBTITLE_ORIGIN_TRANSCRIPTION);
|
||||
assert_eq!(stored["engine"], "whisper");
|
||||
assert_eq!(stored["model"], "ggml-tiny.bin");
|
||||
assert_eq!(stored["language"], "en");
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(rows[0].metadata_json.as_deref()).kind,
|
||||
SubtitleKind::Transcribed
|
||||
);
|
||||
|
||||
assert_eq!(sanitize_model_name("nvidia/parakeet-tdt-0.6b-v3"), "nvidia/parakeet-tdt-0.6b-v3");
|
||||
assert_eq!(sanitize_model_name("phonon-2"), "phonon-2");
|
||||
assert_eq!(sanitize_model_name("C:\\models\\x.bin"), "x.bin");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load diff
650
crates/archivr-core/src/thread_title.rs
Normal file
650
crates/archivr-core/src/thread_title.rs
Normal file
|
|
@ -0,0 +1,650 @@
|
|||
//! On-demand titles for archived X threads.
|
||||
//!
|
||||
//! A user asks for a title from the entry rail; the ordered thread text is sent
|
||||
//! to the selected summary provider with a cheap per-provider model and the
|
||||
//! result is saved as `Thread about <topic> — @author` (just `Thread about
|
||||
//! <topic>` when the author is unknown).
|
||||
//!
|
||||
//! The title model never inherits the summary model (`ARCHIVR_*_MODEL`). It is
|
||||
//! resolved as: the admin's instance setting (passed in by the caller; core
|
||||
//! never reads the auth DB) > `ARCHIVR_ANTHROPIC_TITLE_MODEL` /
|
||||
//! `ARCHIVR_OPENAI_TITLE_MODEL` / `ARCHIVR_CLAUDE_TITLE_MODEL` /
|
||||
//! `ARCHIVR_CODEX_TITLE_MODEL` > a built-in small default. Endpoint, key, CLI
|
||||
//! path and timeout still come from `summarizer::provider_from_env`.
|
||||
//!
|
||||
//! The model returns only the topic phrase; the server builds the rest so the
|
||||
//! title format (prefix and author suffix) is guaranteed regardless of output.
|
||||
|
||||
use anyhow::{Context, Result, bail};
|
||||
use rusqlite::OptionalExtension;
|
||||
use std::path::Path;
|
||||
|
||||
use crate::archive::ArchivePaths;
|
||||
use crate::database;
|
||||
use crate::env_config::optional_env;
|
||||
use crate::summarizer::{self, ProviderConfig};
|
||||
|
||||
pub const TITLE_MAX_TOKENS: u32 = 64;
|
||||
const MAX_TITLE_INPUT_CHARS: usize = 8_000;
|
||||
const MAX_TOPIC_WORDS: usize = 10;
|
||||
const MAX_TOPIC_CHARS: usize = 80;
|
||||
|
||||
const TITLE_SYSTEM_PROMPT: &str = "You name archived X (Twitter) threads for a personal archive index. Reply with ONLY a short topic phrase of 3 to 8 words that completes the sentence 'Thread about …' (for example: migrating a home server to NixOS). Plain text on one line: no quotes, no markdown, no hashtags, no emoji, no @mentions, no trailing punctuation, and do not repeat the words 'Thread about'.";
|
||||
|
||||
const QUOTE_CHARS: &[char] = &['"', '\'', '`', '“', '”', '‘', '’', '«', '»', '*', '_', '#'];
|
||||
const TRAILING_PUNCT: &[char] = &['.', ',', ';', ':', '!', '?', '…'];
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct ThreadTitleInput {
|
||||
pub entry_uid: String,
|
||||
/// Empty when no status JSON names the author.
|
||||
pub author: String,
|
||||
pub content: String,
|
||||
}
|
||||
|
||||
pub fn title_model_env(kind: &str) -> Option<&'static str> {
|
||||
match kind {
|
||||
"anthropic_http" => Some("ARCHIVR_ANTHROPIC_TITLE_MODEL"),
|
||||
"openai_compatible" => Some("ARCHIVR_OPENAI_TITLE_MODEL"),
|
||||
"claude_cli" => Some("ARCHIVR_CLAUDE_TITLE_MODEL"),
|
||||
"codex_cli" => Some("ARCHIVR_CODEX_TITLE_MODEL"),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn default_title_model(kind: &str) -> Option<&'static str> {
|
||||
match kind {
|
||||
"anthropic_http" => Some("claude-haiku-4-5"),
|
||||
"openai_compatible" => Some("gpt-4o-mini"),
|
||||
"claude_cli" => Some("haiku"),
|
||||
"codex_cli" => Some("gpt-6-luna"),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn with_title_model(cfg: ProviderConfig, model: String) -> ProviderConfig {
|
||||
match cfg {
|
||||
ProviderConfig::AnthropicHttp(mut c) => {
|
||||
c.model = model;
|
||||
ProviderConfig::AnthropicHttp(c)
|
||||
}
|
||||
ProviderConfig::OpenAiCompatible(mut c) => {
|
||||
c.model = model;
|
||||
ProviderConfig::OpenAiCompatible(c)
|
||||
}
|
||||
ProviderConfig::ClaudeCli(mut c) => {
|
||||
c.model = Some(model);
|
||||
ProviderConfig::ClaudeCli(c)
|
||||
}
|
||||
ProviderConfig::CodexCli(mut c) => {
|
||||
c.model = Some(model);
|
||||
ProviderConfig::CodexCli(c)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Where an effective title model came from.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum TitleModelSource {
|
||||
Instance,
|
||||
Env,
|
||||
Default,
|
||||
}
|
||||
|
||||
impl TitleModelSource {
|
||||
pub fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Self::Instance => "instance",
|
||||
Self::Env => "env",
|
||||
Self::Default => "default",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Effective title model for `kind`: non-empty trimmed `instance_override` >
|
||||
/// non-empty title env var > built-in default. `None` for unknown kinds.
|
||||
pub fn resolve_title_model(
|
||||
kind: &str,
|
||||
instance_override: Option<&str>,
|
||||
) -> Option<(String, TitleModelSource)> {
|
||||
let (var, default) = (title_model_env(kind)?, default_title_model(kind)?);
|
||||
if let Some(m) = instance_override.map(str::trim).filter(|m| !m.is_empty()) {
|
||||
return Some((m.to_string(), TitleModelSource::Instance));
|
||||
}
|
||||
if let Some(m) = optional_env(var).map(|m| m.trim().to_string()).filter(|m| !m.is_empty()) {
|
||||
return Some((m, TitleModelSource::Env));
|
||||
}
|
||||
Some((default.to_string(), TitleModelSource::Default))
|
||||
}
|
||||
|
||||
/// Provider config for title generation: transport settings from the summary
|
||||
/// env, model from [`resolve_title_model`].
|
||||
pub fn title_provider_from_env(
|
||||
kind: &str,
|
||||
instance_override: Option<&str>,
|
||||
) -> Result<ProviderConfig> {
|
||||
// Validates `kind` and keeps the summary path's missing-key messages.
|
||||
let cfg = summarizer::provider_from_env(kind)?;
|
||||
let Some((model, _)) = resolve_title_model(kind, instance_override) else {
|
||||
bail!("unknown summary provider: {kind}");
|
||||
};
|
||||
Ok(with_title_model(cfg, model))
|
||||
}
|
||||
|
||||
/// Expected, user-facing failure of [`load_thread_title_input`] (entry is not a
|
||||
/// thread, or has no archived text). Anything else is an internal error.
|
||||
#[derive(Debug)]
|
||||
pub struct ThreadTitleUserError(pub String);
|
||||
|
||||
impl std::fmt::Display for ThreadTitleUserError {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str(&self.0)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for ThreadTitleUserError {}
|
||||
|
||||
/// The [`ThreadTitleUserError`] message carried by `error`, if any.
|
||||
pub fn thread_title_user_message(error: &anyhow::Error) -> Option<String> {
|
||||
error.downcast_ref::<ThreadTitleUserError>().map(|m| m.0.clone())
|
||||
}
|
||||
|
||||
/// Loads the thread text and author. `Ok(None)` means the entry does not exist.
|
||||
pub fn load_thread_title_input(
|
||||
paths: &ArchivePaths,
|
||||
entry_uid: &str,
|
||||
) -> Result<Option<ThreadTitleInput>> {
|
||||
let conn = database::open_or_initialize(&paths.archive_path)?;
|
||||
let Some((id, entity_kind, source_metadata_json)) = conn
|
||||
.query_row(
|
||||
"SELECT id, entity_kind, source_metadata_json FROM archived_entries WHERE entry_uid = ?1",
|
||||
[entry_uid],
|
||||
|row| Ok((row.get::<_, i64>(0)?, row.get::<_, String>(1)?, row.get::<_, String>(2)?)),
|
||||
)
|
||||
.optional()?
|
||||
else {
|
||||
return Ok(None);
|
||||
};
|
||||
if entity_kind != "tweet_thread" {
|
||||
return Err(anyhow::Error::new(ThreadTitleUserError(format!(
|
||||
"entry is '{entity_kind}', not an X thread; titles can only be generated for threads"
|
||||
))));
|
||||
}
|
||||
|
||||
let content = summarizer::artifact_text_content(&conn, &paths.store_path, id, &entity_kind)
|
||||
.map_err(|e| {
|
||||
if summarizer::is_unsupported_summary_content_error(&e) {
|
||||
anyhow::Error::new(ThreadTitleUserError(
|
||||
"this thread has no archived text to generate a title from".to_string(),
|
||||
))
|
||||
} else {
|
||||
e
|
||||
}
|
||||
})?;
|
||||
let content: String = content.chars().take(MAX_TITLE_INPUT_CHARS).collect();
|
||||
|
||||
let root_tweet_id = serde_json::from_str::<serde_json::Value>(&source_metadata_json)
|
||||
.ok()
|
||||
.and_then(|v| v["tweet_id"].as_str().map(str::to_string));
|
||||
let author = thread_author(
|
||||
&conn,
|
||||
&paths.store_path,
|
||||
id,
|
||||
root_tweet_id.as_deref(),
|
||||
entry_uid,
|
||||
)?;
|
||||
|
||||
Ok(Some(ThreadTitleInput {
|
||||
entry_uid: entry_uid.to_string(),
|
||||
author,
|
||||
content,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Author of the root status (`tweet-<source_metadata.tweet_id>.json`, as in
|
||||
/// capture's `Thread by @…`), else of the first readable status JSON.
|
||||
fn thread_author(
|
||||
conn: &rusqlite::Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
root_tweet_id: Option<&str>,
|
||||
entry_uid: &str,
|
||||
) -> Result<String> {
|
||||
let root_file = root_tweet_id.map(|id| format!("tweet-{id}.json"));
|
||||
for role in ["raw_tweet_json", "primary_media"] {
|
||||
let mut artifacts: Vec<_> = database::list_entry_artifacts_by_role(conn, entry_id, role)?
|
||||
.into_iter()
|
||||
.filter(|a| a.relpath.ends_with(".json"))
|
||||
.collect();
|
||||
if artifacts.is_empty() {
|
||||
continue;
|
||||
}
|
||||
// Root status first; the rest keep insertion order (stable sort).
|
||||
if let Some(root_file) = root_file.as_deref() {
|
||||
artifacts.sort_by_key(|a| {
|
||||
!(Path::new(&a.relpath).file_name().and_then(|n| n.to_str()) == Some(root_file))
|
||||
});
|
||||
}
|
||||
for artifact in &artifacts {
|
||||
let abs = store_path.join(&artifact.relpath);
|
||||
let parsed = std::fs::read_to_string(&abs)
|
||||
.with_context(|| format!("failed to read {}", abs.display()))
|
||||
.and_then(|raw| {
|
||||
serde_json::from_str::<serde_json::Value>(&raw)
|
||||
.with_context(|| format!("{} is not valid JSON", abs.display()))
|
||||
});
|
||||
let json = match parsed {
|
||||
Ok(json) => json,
|
||||
Err(e) => {
|
||||
eprintln!("warn: thread title {entry_uid}: {e:#}");
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let name = json["author"]["screen_name"]
|
||||
.as_str()
|
||||
.map(|s| s.trim().trim_start_matches('@').trim())
|
||||
.filter(|s| !s.is_empty());
|
||||
if let Some(name) = name {
|
||||
return Ok(name.to_string());
|
||||
}
|
||||
}
|
||||
// Only the first role that has JSON artifacts is consulted, matching
|
||||
// `artifact_text_content`'s legacy `primary_media` fallback.
|
||||
break;
|
||||
}
|
||||
Ok(String::new())
|
||||
}
|
||||
|
||||
pub fn build_title_user_prompt(input: &ThreadTitleInput) -> String {
|
||||
if input.author.is_empty() {
|
||||
format!("Thread:\n{}\n", input.content)
|
||||
} else {
|
||||
format!("Author: @{}\n\nThread:\n{}\n", input.author, input.content)
|
||||
}
|
||||
}
|
||||
|
||||
fn strip_prefix_ci<'a>(s: &'a str, prefix: &str) -> Option<&'a str> {
|
||||
let head = s.get(..prefix.len())?;
|
||||
head.eq_ignore_ascii_case(prefix).then(|| &s[prefix.len()..])
|
||||
}
|
||||
|
||||
fn trim_trailing_punct(s: &str) -> &str {
|
||||
s.trim_end_matches(|c: char| TRAILING_PUNCT.contains(&c) || QUOTE_CHARS.contains(&c))
|
||||
.trim_end()
|
||||
}
|
||||
|
||||
/// Reduces a model reply to a single short topic phrase.
|
||||
pub fn sanitize_topic(raw: &str) -> Result<String> {
|
||||
let line = raw
|
||||
.lines()
|
||||
.filter(|l| !l.trim().starts_with("```"))
|
||||
.map(str::trim)
|
||||
.find(|l| !l.is_empty())
|
||||
.unwrap_or("");
|
||||
let line: String = line.chars().filter(|c| !c.is_control()).collect();
|
||||
|
||||
let mut s: &str = line.trim();
|
||||
for prefix in ["title:", "topic:"] {
|
||||
if let Some(rest) = strip_prefix_ci(s, prefix) {
|
||||
s = rest;
|
||||
break;
|
||||
}
|
||||
}
|
||||
loop {
|
||||
let before = s;
|
||||
s = s.trim().trim_matches(|c: char| QUOTE_CHARS.contains(&c));
|
||||
if let Some(rest) = strip_prefix_ci(s, "thread about ") {
|
||||
s = rest;
|
||||
}
|
||||
if let Some(rest) = strip_prefix_ci(s, "about ") {
|
||||
s = rest;
|
||||
}
|
||||
if s == before {
|
||||
break;
|
||||
}
|
||||
}
|
||||
for sep in [" — @", " – @", " - @"] {
|
||||
if let Some(idx) = s.find(sep) {
|
||||
s = &s[..idx];
|
||||
}
|
||||
}
|
||||
|
||||
let collapsed = s.split_whitespace().collect::<Vec<_>>().join(" ");
|
||||
let trimmed = trim_trailing_punct(&collapsed);
|
||||
let mut topic = trimmed
|
||||
.split(' ')
|
||||
.filter(|w| !w.is_empty())
|
||||
.take(MAX_TOPIC_WORDS)
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
|
||||
if topic.chars().count() > MAX_TOPIC_CHARS {
|
||||
let head: String = topic.chars().take(MAX_TOPIC_CHARS).collect();
|
||||
// `head` is MAX_TOPIC_CHARS chars and the next char exists, so a space
|
||||
// at the cut point is preserved by checking the following char too.
|
||||
let next_is_space = topic.chars().nth(MAX_TOPIC_CHARS) == Some(' ');
|
||||
let cut = if next_is_space {
|
||||
head.as_str()
|
||||
} else {
|
||||
match head.rfind(' ') {
|
||||
Some(idx) => &head[..idx],
|
||||
None => head.as_str(),
|
||||
}
|
||||
};
|
||||
topic = trim_trailing_punct(cut).to_string();
|
||||
}
|
||||
|
||||
if topic.is_empty() {
|
||||
bail!("title provider returned no usable title");
|
||||
}
|
||||
Ok(topic)
|
||||
}
|
||||
|
||||
/// `author` is empty when unknown; the ` — @…` suffix is then omitted.
|
||||
pub fn format_thread_title(topic: &str, author: &str) -> String {
|
||||
if author.is_empty() {
|
||||
format!("Thread about {topic}")
|
||||
} else {
|
||||
format!("Thread about {topic} — @{author}")
|
||||
}
|
||||
}
|
||||
|
||||
fn short(s: &str) -> String {
|
||||
s.chars().take(200).collect()
|
||||
}
|
||||
|
||||
/// Asks the provider for a topic and builds the final title.
|
||||
pub fn generate_thread_title(cfg: &ProviderConfig, input: &ThreadTitleInput) -> Result<String> {
|
||||
let out = summarizer::complete_plain(
|
||||
cfg,
|
||||
TITLE_SYSTEM_PROMPT,
|
||||
&build_title_user_prompt(input),
|
||||
TITLE_MAX_TOKENS,
|
||||
)
|
||||
.context("title generation failed")?;
|
||||
let topic =
|
||||
sanitize_topic(&out.text).with_context(|| format!("raw reply: {}", short(&out.text)))?;
|
||||
Ok(format_thread_title(&topic, &input.author))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::summarizer::{CliProviderConfig, HttpProviderConfig};
|
||||
|
||||
#[test]
|
||||
fn sanitize_topic_cleans_model_replies() {
|
||||
let ok = |raw: &str| sanitize_topic(raw).unwrap();
|
||||
assert_eq!(ok("\"Rust async runtimes compared.\""), "Rust async runtimes compared");
|
||||
assert_eq!(ok("Thread about NixOS on a Pi"), "NixOS on a Pi");
|
||||
assert_eq!(ok("```\nTitle: **Home lab networking**\n```"), "Home lab networking");
|
||||
assert_eq!(ok("\nFirst line topic\nSecond line"), "First line topic");
|
||||
assert_eq!(ok("Foo bar — @alice"), "Foo bar");
|
||||
let twenty = (1..=20).map(|i| format!("w{i}")).collect::<Vec<_>>().join(" ");
|
||||
assert_eq!(ok(&twenty), "w1 w2 w3 w4 w5 w6 w7 w8 w9 w10");
|
||||
let long_word = "x".repeat(120);
|
||||
assert!(ok(&long_word).chars().count() <= MAX_TOPIC_CHARS);
|
||||
let long_words = vec!["abcdefghijk"; 10].join(" ");
|
||||
let cut = ok(&long_words);
|
||||
assert!(cut.chars().count() <= MAX_TOPIC_CHARS);
|
||||
assert!(!cut.ends_with(' '));
|
||||
assert!(sanitize_topic(" ").is_err());
|
||||
assert!(sanitize_topic("\"\"").is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_thread_title_uses_em_dash_suffix() {
|
||||
assert_eq!(format_thread_title("x y", "bob"), "Thread about x y — @bob");
|
||||
assert_eq!(format_thread_title("x y", ""), "Thread about x y");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn title_models_cover_all_provider_kinds() {
|
||||
for kind in summarizer::PROVIDER_KINDS {
|
||||
assert!(title_model_env(kind).is_some(), "{kind}");
|
||||
assert!(default_title_model(kind).is_some(), "{kind}");
|
||||
}
|
||||
assert_eq!(title_model_env("claude_cli"), Some("ARCHIVR_CLAUDE_TITLE_MODEL"));
|
||||
assert_eq!(default_title_model("codex_cli"), Some("gpt-6-luna"));
|
||||
assert_eq!(default_title_model("anthropic_http"), Some("claude-haiku-4-5"));
|
||||
assert_eq!(title_model_env("gemini"), None);
|
||||
assert_eq!(default_title_model("gemini"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn with_title_model_sets_model_on_every_variant() {
|
||||
let http = HttpProviderConfig {
|
||||
endpoint: "https://example.invalid".into(),
|
||||
api_key: "k".into(),
|
||||
model: "big".into(),
|
||||
timeout_secs: 1,
|
||||
};
|
||||
let cli = CliProviderConfig {
|
||||
executable: "claude".into(),
|
||||
model: None,
|
||||
timeout_secs: 1,
|
||||
};
|
||||
let m = || "small".to_string();
|
||||
match with_title_model(ProviderConfig::AnthropicHttp(http.clone()), m()) {
|
||||
ProviderConfig::AnthropicHttp(c) => assert_eq!(c.model, "small"),
|
||||
other => panic!("{other:?}"),
|
||||
}
|
||||
match with_title_model(ProviderConfig::OpenAiCompatible(http), m()) {
|
||||
ProviderConfig::OpenAiCompatible(c) => assert_eq!(c.model, "small"),
|
||||
other => panic!("{other:?}"),
|
||||
}
|
||||
match with_title_model(ProviderConfig::ClaudeCli(cli.clone()), m()) {
|
||||
ProviderConfig::ClaudeCli(c) => assert_eq!(c.model.as_deref(), Some("small")),
|
||||
other => panic!("{other:?}"),
|
||||
}
|
||||
match with_title_model(ProviderConfig::CodexCli(cli), m()) {
|
||||
ProviderConfig::CodexCli(c) => assert_eq!(c.model.as_deref(), Some("small")),
|
||||
other => panic!("{other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Archive with one entry of `entity_kind` and the given raw tweet JSON files.
|
||||
fn fixture(
|
||||
entity_kind: &str,
|
||||
tweets: &[(&str, serde_json::Value)],
|
||||
) -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let paths = crate::archive::initialize_archive(
|
||||
temp.path(),
|
||||
&temp.path().join("store"),
|
||||
"Test archive",
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
let user_id = database::ensure_default_user(&conn).unwrap();
|
||||
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
|
||||
let source_id = database::upsert_source_identity(
|
||||
&conn,
|
||||
"x",
|
||||
entity_kind,
|
||||
Some("9001"),
|
||||
Some("https://x.com/alice/status/9001"),
|
||||
"x:thread:9001",
|
||||
)
|
||||
.unwrap();
|
||||
let entry = database::create_archived_entry(
|
||||
&conn,
|
||||
&database::NewEntry {
|
||||
source_identity_id: source_id,
|
||||
archive_run_id: run.id,
|
||||
parent_entry_id: None,
|
||||
root_entry_id: None,
|
||||
created_by_user_id: user_id,
|
||||
owned_by_user_id: user_id,
|
||||
source_kind: "x".to_string(),
|
||||
entity_kind: entity_kind.to_string(),
|
||||
title: Some("Thread by @alice".to_string()),
|
||||
visibility: "private".to_string(),
|
||||
representation_kind: entity_kind.to_string(),
|
||||
source_metadata_json: r#"{"tweet_id":"9001"}"#.to_string(),
|
||||
display_metadata_json: None,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::create_dir_all(paths.store_path.join("raw_tweets")).unwrap();
|
||||
for (relpath, body) in tweets {
|
||||
std::fs::write(paths.store_path.join(relpath), body.to_string()).unwrap();
|
||||
database::add_entry_artifact(
|
||||
&conn,
|
||||
&database::NewArtifact {
|
||||
entry_id: entry.id,
|
||||
artifact_role: "raw_tweet_json".to_string(),
|
||||
storage_area: "raw_tweets".to_string(),
|
||||
relpath: relpath.to_string(),
|
||||
blob_id: None,
|
||||
logical_path: None,
|
||||
metadata_json: None,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
(temp, paths, entry)
|
||||
}
|
||||
|
||||
fn alice_thread() -> Vec<(&'static str, serde_json::Value)> {
|
||||
vec![
|
||||
(
|
||||
"raw_tweets/tweet-9001.json",
|
||||
serde_json::json!({
|
||||
"full_text": "1/ Comparing Rust async runtimes.",
|
||||
"author": { "screen_name": "@alice" }
|
||||
}),
|
||||
),
|
||||
(
|
||||
"raw_tweets/tweet-9002.json",
|
||||
serde_json::json!({
|
||||
"full_text": "2/ Tokio wins on ecosystem.",
|
||||
"author": { "screen_name": "alice" }
|
||||
}),
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_thread_title_input_reads_thread_text_and_author() {
|
||||
let (_temp, paths, entry) = fixture("tweet_thread", &alice_thread());
|
||||
let input = load_thread_title_input(&paths, &entry.entry_uid)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
assert_eq!(input.entry_uid, entry.entry_uid);
|
||||
assert_eq!(input.author, "alice");
|
||||
assert!(
|
||||
input
|
||||
.content
|
||||
.contains("1/ Comparing Rust async runtimes.\n\n---\n\n2/ Tokio wins on ecosystem."),
|
||||
"{}",
|
||||
input.content
|
||||
);
|
||||
assert!(!input.content.contains("Thread by @alice"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_thread_title_input_prefers_root_status_author() {
|
||||
let tweets = vec![
|
||||
(
|
||||
"raw_tweets/tweet-8000.json",
|
||||
serde_json::json!({ "full_text": "quoted", "author": { "screen_name": "bob" } }),
|
||||
),
|
||||
(
|
||||
"raw_tweets/tweet-9001.json",
|
||||
serde_json::json!({ "full_text": "root", "author": { "screen_name": "alice" } }),
|
||||
),
|
||||
];
|
||||
let (_temp, paths, entry) = fixture("tweet_thread", &tweets);
|
||||
let input = load_thread_title_input(&paths, &entry.entry_uid)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
assert_eq!(input.author, "alice");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_thread_title_input_rejects_non_threads_and_empty_threads() {
|
||||
let (_temp, paths, entry) = fixture("page", &[]);
|
||||
let err = load_thread_title_input(&paths, &entry.entry_uid).unwrap_err();
|
||||
assert!(format!("{err:#}").contains("not an X thread"), "{err:#}");
|
||||
assert!(thread_title_user_message(&err).is_some());
|
||||
assert!(load_thread_title_input(&paths, "no-such-uid").unwrap().is_none());
|
||||
|
||||
let empty = vec![(
|
||||
"raw_tweets/tweet-9001.json",
|
||||
serde_json::json!({ "full_text": "", "author": { "screen_name": "alice" } }),
|
||||
)];
|
||||
let (_temp2, paths2, entry2) = fixture("tweet_thread", &empty);
|
||||
let err = load_thread_title_input(&paths2, &entry2.entry_uid).unwrap_err();
|
||||
assert!(format!("{err:#}").contains("no archived text"), "{err:#}");
|
||||
assert!(thread_title_user_message(&err).is_some());
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn generate_thread_title_runs_claude_cli_with_title_model() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let script = dir.path().join("fake-claude");
|
||||
crate::downloader::write_script(
|
||||
&script,
|
||||
"#!/bin/sh\ncat >/dev/null\ncase \" $* \" in *\" --model haiku \"*) ;; *) echo \"bad args: $*\" >&2; exit 3;; esac\nprintf '%s\\n' '\"Rust async runtimes compared.\"'\n",
|
||||
);
|
||||
let cfg = ProviderConfig::ClaudeCli(CliProviderConfig {
|
||||
executable: script,
|
||||
model: Some("haiku".into()),
|
||||
timeout_secs: 30,
|
||||
});
|
||||
let input = ThreadTitleInput {
|
||||
entry_uid: "uid".into(),
|
||||
author: "alice".into(),
|
||||
content: "1/ Comparing Rust async runtimes.".into(),
|
||||
};
|
||||
// Parallel tests forking while the script fd was open can briefly make
|
||||
// exec fail with ETXTBSY (rust-lang/rust#114554); retry that case only.
|
||||
let mut attempt = 0;
|
||||
let title = loop {
|
||||
match generate_thread_title(&cfg, &input) {
|
||||
Err(e) if attempt < 20 && format!("{e:#}").contains("Text file busy") => {
|
||||
attempt += 1;
|
||||
std::thread::sleep(std::time::Duration::from_millis(50));
|
||||
}
|
||||
other => break other.unwrap(),
|
||||
}
|
||||
};
|
||||
assert_eq!(title, "Thread about Rust async runtimes compared — @alice");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resolve_title_model_prefers_instance_then_env_then_default() {
|
||||
// Only this test touches the codex title env var; restored below.
|
||||
let var = "ARCHIVR_CODEX_TITLE_MODEL";
|
||||
let previous = std::env::var_os(var);
|
||||
unsafe { std::env::remove_var(var) };
|
||||
assert_eq!(
|
||||
resolve_title_model("codex_cli", None),
|
||||
Some(("gpt-6-luna".into(), TitleModelSource::Default))
|
||||
);
|
||||
assert_eq!(
|
||||
resolve_title_model("codex_cli", Some(" ")),
|
||||
Some(("gpt-6-luna".into(), TitleModelSource::Default))
|
||||
);
|
||||
unsafe { std::env::set_var(var, " env-model ") };
|
||||
assert_eq!(
|
||||
resolve_title_model("codex_cli", None),
|
||||
Some(("env-model".into(), TitleModelSource::Env))
|
||||
);
|
||||
assert_eq!(
|
||||
resolve_title_model("codex_cli", Some(" inst-model ")),
|
||||
Some(("inst-model".into(), TitleModelSource::Instance))
|
||||
);
|
||||
unsafe {
|
||||
match previous {
|
||||
Some(v) => std::env::set_var(var, v),
|
||||
None => std::env::remove_var(var),
|
||||
}
|
||||
}
|
||||
assert_eq!(resolve_title_model("gemini", Some("x")), None);
|
||||
assert_eq!(TitleModelSource::Env.as_str(), "env");
|
||||
}
|
||||
}
|
||||
1742
crates/archivr-core/src/transcriber.rs
Normal file
1742
crates/archivr-core/src/transcriber.rs
Normal file
File diff suppressed because it is too large
Load diff
Loading…
Add table
Add a link
Reference in a new issue