//! Per-entry LLM summaries. //! //! A summary is a *regenerable child record* of an entry (`entry_summaries`), //! not a column on `archived_entries` and not an artifact on disk: an entry can //! carry several summaries (one per provider / model / prompt version), any of //! them can be discarded and recomputed, and none of them is part of the //! preserved capture. Generation is manual-only — nothing in `capture.rs` calls //! into this module. //! //! Four providers sit behind one [`SummaryProvider`] trait: two HTTP APIs and //! two local CLIs. They are configured by environment variable, never by TOML, //! matching how the rest of the tree resolves external tools (`ARCHIVR_YT_DLP`, //! `ARCHIVR_SINGLE_FILE`, `ARCHIVR_TWEET_SCRAPER`, …) and keeping API keys out //! of any file the archive would otherwise persist. use anyhow::{Context, Result, anyhow, bail}; use std::{ env, io::Write, path::{Path, PathBuf}, process::{Command, Stdio}, sync::mpsc, thread, time::Duration, }; use crate::{archive::ArchivePaths, database, hash}; /// Bump whenever the prompt text below changes in a way that would produce a /// materially different summary. It is part of the `entry_summaries` cache key, /// so a bump makes every stored summary regenerate on next request instead of /// silently mixing outputs from two different prompts. pub const PROMPT_VERSION: &str = "v1-2026-08-22"; /// Upper bound on characters fed to a model. Archived pages run to hundreds of /// kilobytes; past this point we are paying for tokens that do not change a /// five-sentence summary. Truncation happens *before* hashing so the cache key /// describes exactly what the model saw. const MAX_INPUT_CHARS: usize = 48_000; const DEFAULT_HTTP_TIMEOUT_SECS: u64 = 120; const DEFAULT_CLI_TIMEOUT_SECS: u64 = 300; /// The instruction half of the prompt. JSON output is requested because parsing /// prose out of a free-form answer is the single most fragile part of an LLM /// integration; a JSON object survives models that like to add pleasantries. const SYSTEM_PROMPT: &str = "\ You summarize archived web content for a personal archive index. Reply with a single JSON object and nothing else — no markdown fence, no prose before or after. The object has exactly these keys: \"tldr\": one sentence, at most 25 words. \"summary\": 4 to 6 sentences of plain English describing what the content says, its claims, and its conclusion. No preamble like \"This article discusses\". \"tags\": an array of at most 5 short lowercase topic tags. Write in English regardless of the source language. If the content is too short or empty to summarize, still return the object and say so in \"summary\"."; // ── Request / output types ───────────────────────────────────────────────── /// Everything the prompt builder needs about one entry. #[derive(Debug, Clone, PartialEq, Eq)] pub struct SummaryRequest { pub entry_uid: String, pub title: Option, pub source_kind: String, pub entity_kind: String, pub content: String, } /// What a provider produced. `model` is echoed back because HTTP providers may /// resolve an alias (`claude-3-5-sonnet-latest`) to a dated concrete model, and /// the concrete one is what we want recorded against the summary. #[derive(Debug, Clone, PartialEq, Eq)] pub struct SummaryOutput { pub text: String, pub model: Option, } /// One way of turning a [`SummaryRequest`] into text. /// /// `Send + Sync` so a boxed provider can cross into the server's /// `spawn_blocking` worker. pub trait SummaryProvider: Send + Sync { /// Stable identifier persisted as `entry_summaries.provider_kind`. fn kind(&self) -> &'static str; fn model(&self) -> Option<&str>; fn summarize(&self, request: &SummaryRequest) -> Result; } // ── Configuration ────────────────────────────────────────────────────────── #[derive(Debug, Clone, PartialEq, Eq)] pub struct HttpProviderConfig { /// Full URL, e.g. `https://api.anthropic.com/v1/messages`. pub endpoint: String, pub api_key: String, pub model: String, pub timeout_secs: u64, } #[derive(Debug, Clone, PartialEq, Eq)] pub struct CliProviderConfig { /// Resolved via `ARCHIVR_CLAUDE_CLI` / `ARCHIVR_CODEX_CLI`; a bare name is /// left for the OS to resolve on `PATH`, as elsewhere in the tree. pub executable: PathBuf, pub model: Option, pub timeout_secs: u64, } #[derive(Debug, Clone, PartialEq, Eq)] pub enum ProviderConfig { AnthropicHttp(HttpProviderConfig), OpenAiCompatible(HttpProviderConfig), ClaudeCli(CliProviderConfig), CodexCli(CliProviderConfig), } pub const PROVIDER_KINDS: [&str; 4] = [ "anthropic_http", "openai_compatible", "claude_cli", "codex_cli", ]; pub fn provider_from_config(cfg: ProviderConfig) -> Box { match cfg { ProviderConfig::AnthropicHttp(c) => Box::new(AnthropicHttpProvider(c)), ProviderConfig::OpenAiCompatible(c) => Box::new(OpenAiCompatibleProvider(c)), ProviderConfig::ClaudeCli(c) => Box::new(ClaudeCliProvider(c)), ProviderConfig::CodexCli(c) => Box::new(CodexCliProvider(c)), } } /// Reads a required env var, failing with the *exact variable name* so the /// server can hand a caller an actionable 400 rather than "not configured". fn required_env(name: &str) -> Result { match env::var(name) { Ok(v) if !v.trim().is_empty() => Ok(v), _ => bail!("missing required environment variable: {name}"), } } fn env_or(name: &str, default: &str) -> String { env::var(name) .ok() .filter(|v| !v.trim().is_empty()) .unwrap_or_else(|| default.to_string()) } fn optional_env(name: &str) -> Option { env::var(name).ok().filter(|v| !v.trim().is_empty()) } fn env_timeout(name: &str, default: u64) -> u64 { env::var(name) .ok() .and_then(|v| v.trim().parse::().ok()) .filter(|v| *v > 0) .unwrap_or(default) } /// Resolve a CLI executable path. /// /// Priority: `env_name` override → first `well_known_absolute` path that /// exists → `HOME/.local/bin/` if it exists → bare name (relies on the /// server's PATH). The macOS defaults matter for `codex`, which the ChatGPT /// desktop app installs at `/Applications/ChatGPT.app/Contents/Resources/codex` /// and does not add to PATH. fn resolve_cli(env_name: &str, well_known_absolute: &[&str], bare: &str) -> PathBuf { if let Some(explicit) = optional_env(env_name) { return PathBuf::from(explicit); } for candidate in well_known_absolute { let p = Path::new(candidate); if p.is_file() { return p.to_path_buf(); } } if let Some(home) = env::var_os("HOME") { let mut p = PathBuf::from(home); p.push(".local/bin"); p.push(bare); if p.is_file() { return p; } } PathBuf::from(bare) } /// Builds a provider configuration for `kind` purely from the environment. pub fn provider_from_env(kind: &str) -> Result { match kind { "anthropic_http" => Ok(ProviderConfig::AnthropicHttp(HttpProviderConfig { endpoint: env_or("ARCHIVR_ANTHROPIC_URL", "https://api.anthropic.com/v1/messages"), api_key: required_env("ARCHIVR_ANTHROPIC_API_KEY")?, model: env_or("ARCHIVR_ANTHROPIC_MODEL", "claude-3-5-sonnet-latest"), timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS), })), "openai_compatible" => Ok(ProviderConfig::OpenAiCompatible(HttpProviderConfig { endpoint: env_or("ARCHIVR_OPENAI_URL", "https://api.openai.com/v1/chat/completions"), api_key: required_env("ARCHIVR_OPENAI_API_KEY")?, model: env_or("ARCHIVR_OPENAI_MODEL", "gpt-4o-mini"), timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS), })), "claude_cli" => Ok(ProviderConfig::ClaudeCli(CliProviderConfig { executable: resolve_cli( "ARCHIVR_CLAUDE_CLI", &["/opt/homebrew/bin/claude", "/usr/local/bin/claude"], "claude", ), model: optional_env("ARCHIVR_CLAUDE_MODEL"), timeout_secs: env_timeout("ARCHIVR_SUMMARY_CLI_TIMEOUT", DEFAULT_CLI_TIMEOUT_SECS), })), "codex_cli" => Ok(ProviderConfig::CodexCli(CliProviderConfig { executable: resolve_cli( "ARCHIVR_CODEX_CLI", &[ "/Applications/ChatGPT.app/Contents/Resources/codex", "/opt/homebrew/bin/codex", "/usr/local/bin/codex", ], "codex", ), model: optional_env("ARCHIVR_CODEX_MODEL"), timeout_secs: env_timeout("ARCHIVR_SUMMARY_CLI_TIMEOUT", DEFAULT_CLI_TIMEOUT_SECS), })), other => bail!( "unknown summary provider: {other} (expected one of {})", PROVIDER_KINDS.join(", ") ), } } // ── Prompt assembly ──────────────────────────────────────────────────────── /// The user half of the prompt: entry metadata as a small header, then content. pub fn build_user_prompt(request: &SummaryRequest) -> String { let mut s = String::new(); if let Some(title) = request.title.as_deref().filter(|t| !t.trim().is_empty()) { s.push_str(&format!("Title: {title}\n")); } s.push_str(&format!( "Source: {} / {}\n\nContent:\n{}\n", request.source_kind, request.entity_kind, request.content )); s } /// CLIs take a single prompt string on stdin, so the system half is prepended /// rather than passed as a separate role. fn build_combined_prompt(request: &SummaryRequest) -> String { format!("{SYSTEM_PROMPT}\n\n---\n\n{}", build_user_prompt(request)) } // ── HTTP providers ───────────────────────────────────────────────────────── fn http_client(timeout_secs: u64) -> Result { reqwest::blocking::Client::builder() // reqwest's own timeout covers connect + read, which is all a // request/response provider needs — no watchdog thread required. .timeout(Duration::from_secs(timeout_secs)) .build() .context("failed to build HTTP client for summary provider") } /// Body builder kept separate from the transport so it can be unit-tested /// without a network round-trip. pub fn anthropic_request_body(model: &str, request: &SummaryRequest) -> serde_json::Value { serde_json::json!({ "model": model, "max_tokens": 1024, "messages": [{ "role": "user", "content": build_combined_prompt(request), }], }) } pub fn openai_request_body(model: &str, request: &SummaryRequest) -> serde_json::Value { serde_json::json!({ "model": model, "messages": [ { "role": "system", "content": SYSTEM_PROMPT }, { "role": "user", "content": build_user_prompt(request) }, ], }) } struct AnthropicHttpProvider(HttpProviderConfig); impl SummaryProvider for AnthropicHttpProvider { fn kind(&self) -> &'static str { "anthropic_http" } fn model(&self) -> Option<&str> { Some(&self.0.model) } fn summarize(&self, request: &SummaryRequest) -> Result { let body = anthropic_request_body(&self.0.model, request); let resp = http_client(self.0.timeout_secs)? .post(&self.0.endpoint) .header("x-api-key", &self.0.api_key) .header("anthropic-version", "2023-06-01") .header("content-type", "application/json") .body(body.to_string()) .send() .with_context(|| format!("request to {} failed", self.0.endpoint))?; let status = resp.status(); let text = resp.text().unwrap_or_default(); if !status.is_success() { bail!("anthropic API returned {status}: {}", truncate_for_error(&text)); } parse_anthropic_response(&text) } } pub fn parse_anthropic_response(body: &str) -> Result { let json: serde_json::Value = serde_json::from_str(body).context("anthropic response was not JSON")?; let text = json["content"][0]["text"] .as_str() .ok_or_else(|| anyhow!("anthropic response had no content[0].text"))?; Ok(SummaryOutput { text: text.to_string(), model: json["model"].as_str().map(str::to_string), }) } struct OpenAiCompatibleProvider(HttpProviderConfig); impl SummaryProvider for OpenAiCompatibleProvider { fn kind(&self) -> &'static str { "openai_compatible" } fn model(&self) -> Option<&str> { Some(&self.0.model) } fn summarize(&self, request: &SummaryRequest) -> Result { let body = openai_request_body(&self.0.model, request); let resp = http_client(self.0.timeout_secs)? .post(&self.0.endpoint) .header("authorization", format!("Bearer {}", self.0.api_key)) .header("content-type", "application/json") .body(body.to_string()) .send() .with_context(|| format!("request to {} failed", self.0.endpoint))?; let status = resp.status(); let text = resp.text().unwrap_or_default(); if !status.is_success() { bail!( "openai-compatible API returned {status}: {}", truncate_for_error(&text) ); } parse_openai_response(&text) } } pub fn parse_openai_response(body: &str) -> Result { let json: serde_json::Value = serde_json::from_str(body).context("openai-compatible response was not JSON")?; let text = json["choices"][0]["message"]["content"] .as_str() .ok_or_else(|| anyhow!("response had no choices[0].message.content"))?; Ok(SummaryOutput { text: text.to_string(), model: json["model"].as_str().map(str::to_string), }) } fn truncate_for_error(s: &str) -> String { let trimmed = s.trim(); if trimmed.chars().count() <= 400 { return trimmed.to_string(); } trimmed.chars().take(400).collect::() + "…" } // ── CLI providers ────────────────────────────────────────────────────────── /// Runs `executable args…`, writes `prompt` to its stdin, and returns stdout. /// /// `archivr-core` deliberately has no async runtime and the tree carries no /// `wait_timeout` dependency, so the timeout is enforced by structure rather /// than by a library: stdout is drained on its own thread and handed back over /// a channel, which leaves the calling thread free to `recv_timeout` and kill /// the child if it overruns. stdin is written on a third thread because a /// 48 KB prompt can exceed the pipe buffer, and writing it inline would /// deadlock against a child that is waiting for us to read its output. fn run_cli( executable: &Path, args: &[&str], prompt: &str, timeout_secs: u64, ) -> Result { let mut child = Command::new(executable) .args(args) .stdin(Stdio::piped()) .stdout(Stdio::piped()) .stderr(Stdio::piped()) .spawn() .with_context(|| format!("failed to spawn {}", executable.display()))?; let mut stdin = child .stdin .take() .ok_or_else(|| anyhow!("failed to open stdin for {}", executable.display()))?; let prompt_owned = prompt.to_string(); thread::spawn(move || { let _ = stdin.write_all(prompt_owned.as_bytes()); // Dropping stdin closes the pipe, which is what tells the CLI the // prompt is complete. }); let stdout = child .stdout .take() .ok_or_else(|| anyhow!("failed to open stdout for {}", executable.display()))?; let (tx, rx) = mpsc::channel(); thread::spawn(move || { let mut buf = String::new(); use std::io::Read; let mut stdout = stdout; let res = stdout.read_to_string(&mut buf).map(|_| buf); let _ = tx.send(res); }); let collected = match rx.recv_timeout(Duration::from_secs(timeout_secs)) { Ok(res) => res.with_context(|| format!("failed to read stdout of {}", executable.display()))?, Err(_) => { let _ = child.kill(); let _ = child.wait(); bail!( "{} timed out after {timeout_secs}s", executable.display() ); } }; let status = child .wait() .with_context(|| format!("failed to wait for {}", executable.display()))?; if !status.success() { let mut stderr = String::new(); if let Some(mut e) = child.stderr.take() { use std::io::Read; let _ = e.read_to_string(&mut stderr); } bail!( "{} exited with {status}: {}", executable.display(), truncate_for_error(&stderr) ); } Ok(collected) } struct ClaudeCliProvider(CliProviderConfig); impl SummaryProvider for ClaudeCliProvider { fn kind(&self) -> &'static str { "claude_cli" } fn model(&self) -> Option<&str> { self.0.model.as_deref() } fn summarize(&self, request: &SummaryRequest) -> Result { // `claude -p --output-format text` is the documented one-shot // ("print") mode of the Claude Code CLI: it reads the prompt from // stdin, writes the answer to stdout, and exits. let mut args: Vec<&str> = vec!["-p", "--output-format", "text"]; if let Some(model) = self.0.model.as_deref() { args.push("--model"); args.push(model); } let out = run_cli(&self.0.executable, &args, &build_combined_prompt(request), self.0.timeout_secs)?; Ok(SummaryOutput { text: out, model: self.0.model.clone(), }) } } /// Codex invocation lives in its own module because its one-shot interface is /// the least stable of the four. /// /// Primary form is `codex exec --output-last-message -`, which reads the /// prompt from stdin and writes ONLY the final assistant message to ``. /// Without `--output-last-message`, stdout is polluted with a header /// (`OpenAI Codex vX`, session id, model, sandbox, …) and a footer /// (`tokens used`, message replay), and the JSON extractor can pick up the /// echoed user prompt instead of the real answer. Older builds that reject /// `-` as stdin marker fall back to a positional prompt. mod codex { use super::*; /// A short-lived path in the OS temp dir. Unique per (pid, wall time) so /// concurrent summarizations don't collide. fn last_message_temp_path() -> PathBuf { let stamp = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map(|d| d.as_nanos()) .unwrap_or_default(); env::temp_dir().join(format!( "archivr-codex-{}-{}.txt", std::process::id(), stamp )) } fn missing_binary_hint(cfg: &CliProviderConfig) -> &'static str { // Only nudge users about the env var when we're running the default // bare "codex" and it failed — an explicit ARCHIVR_CODEX_CLI path // failure is their configuration, not a discovery gap. if cfg.executable == Path::new("codex") { " (hint: set ARCHIVR_CODEX_CLI to your codex binary; on macOS the ChatGPT desktop app installs it at /Applications/ChatGPT.app/Contents/Resources/codex)" } else { "" } } fn read_and_cleanup(path: &Path) -> Option { let out = std::fs::read_to_string(path).ok()?; let _ = std::fs::remove_file(path); let trimmed = out.trim(); if trimmed.is_empty() { None } else { Some(trimmed.to_string()) } } pub fn run(cfg: &CliProviderConfig, prompt: &str) -> Result { let out_path = last_message_temp_path(); let out_str = out_path.to_string_lossy().into_owned(); // Primary: stdin prompt + --output-last-message. let mut primary: Vec = vec![ "exec".into(), "--output-last-message".into(), out_str.clone(), ]; if let Some(model) = cfg.model.as_deref() { primary.push("--model".into()); primary.push(model.into()); } primary.push("-".into()); let primary_refs: Vec<&str> = primary.iter().map(String::as_str).collect(); let primary_err = match run_cli(&cfg.executable, &primary_refs, prompt, cfg.timeout_secs) { Ok(_) => { if let Some(text) = read_and_cleanup(&out_path) { return Ok(text); } // Codex succeeded but wrote nothing to the file — extremely // rare, but treat as a soft failure so we try the fallback. anyhow!("codex produced no last-message output") } Err(e) => { let _ = std::fs::remove_file(&out_path); e } }; // Fallback: positional prompt, no stdin, same --output-last-message. let mut fb: Vec = vec![ "exec".into(), "--output-last-message".into(), out_str.clone(), ]; if let Some(model) = cfg.model.as_deref() { fb.push("--model".into()); fb.push(model.into()); } fb.push(prompt.into()); let out = Command::new(&cfg.executable) .args(&fb) .output() .with_context(|| { format!( "codex `exec -` failed ({primary_err:#}); positional fallback also failed to spawn{}", missing_binary_hint(cfg) ) })?; if !out.status.success() { let _ = std::fs::remove_file(&out_path); bail!( "codex `exec -` failed ({primary_err:#}); positional fallback exited with {}: {}", out.status, truncate_for_error(&String::from_utf8_lossy(&out.stderr)) ); } if let Some(text) = read_and_cleanup(&out_path) { return Ok(text); } // Last resort — the child succeeded but wrote nothing to the file. Fall // back to raw stdout so the caller has *something* to normalize. Ok(String::from_utf8_lossy(&out.stdout).to_string()) } } struct CodexCliProvider(CliProviderConfig); impl SummaryProvider for CodexCliProvider { fn kind(&self) -> &'static str { "codex_cli" } fn model(&self) -> Option<&str> { self.0.model.as_deref() } fn summarize(&self, request: &SummaryRequest) -> Result { let out = codex::run(&self.0, &build_combined_prompt(request))?; Ok(SummaryOutput { text: out, model: self.0.model.clone(), }) } } // ── Content extraction ───────────────────────────────────────────────────── /// Strips markup from an archived HTML page. /// /// Deliberately regex-based rather than a real parser: `html5ever` is not in /// the dependency tree, and pulling a full HTML parser in to feed a language /// model — which tolerates imperfect whitespace and stray angle brackets /// fine — is not worth the build cost. `\

Hello world

"; let text = strip_html(html); assert!(text.contains("Hello world")); assert!(!text.contains("color:red")); assert!(!text.contains("var x")); } #[test] fn strip_html_keeps_paragraphs_apart() { let text = strip_html("

One

Two

"); // Without block-level break handling these would run together as // "OneTwo", which reads as a single garbled sentence to the model. assert!(text.contains("One")); assert!(text.contains("Two")); assert!(!text.contains("OneTwo")); } #[test] fn strip_html_decodes_common_entities() { assert_eq!(strip_html("

a & b  c

").replace('\u{a0}', " "), "a & b c"); } #[test] fn extract_tweet_text_handles_flat_and_wrapped_shapes() { let flat = serde_json::json!({ "full_text": "tweet body" }); assert_eq!(extract_tweet_text(&flat).unwrap(), "tweet body"); let wrapped = serde_json::json!({ "tweet": { "text": "wrapped body" } }); assert_eq!(extract_tweet_text(&wrapped).unwrap(), "wrapped body"); let threaded = serde_json::json!({ "full_text": "first", "thread": [{ "full_text": "second" }], }); assert_eq!(extract_tweet_text(&threaded).unwrap(), "first\n\nsecond"); assert!(extract_tweet_text(&serde_json::json!({ "id": 1 })).is_none()); } // ── Output normalization ─────────────────────────────────────────────── #[test] fn normalize_summary_json_passes_through_clean_json() { let raw = r#"{"tldr":"t","summary":"s","tags":["a"]}"#; let v: serde_json::Value = serde_json::from_str(&normalize_summary_json(raw)).unwrap(); assert_eq!(v["tldr"], "t"); assert_eq!(v["tags"][0], "a"); } #[test] fn normalize_summary_json_strips_markdown_fences() { let raw = "```json\n{\"tldr\":\"t\",\"summary\":\"s\",\"tags\":[]}\n```"; let v: serde_json::Value = serde_json::from_str(&normalize_summary_json(raw)).unwrap(); assert_eq!(v["summary"], "s"); } #[test] fn normalize_summary_json_recovers_json_wrapped_in_prose() { let raw = "Sure! Here you go:\n{\"tldr\":\"t\",\"summary\":\"s\",\"tags\":[]}\nHope that helps."; let v: serde_json::Value = serde_json::from_str(&normalize_summary_json(raw)).unwrap(); assert_eq!(v["tldr"], "t"); } #[test] fn normalize_summary_json_wraps_unparseable_output_rather_than_losing_it() { // A model that ignored the format instruction still produced something // a human can read; discarding it would be worse than a missing tldr. let v: serde_json::Value = serde_json::from_str(&normalize_summary_json("just prose")).unwrap(); assert_eq!(v["summary"], "just prose"); assert_eq!(v["tldr"], ""); } // ── CLI runner ───────────────────────────────────────────────────────── #[test] fn run_cli_round_trips_stdin_to_stdout() { // `cat` stands in for a provider CLI: it proves the prompt reaches the // child's stdin and the child's stdout comes back intact. let out = run_cli(Path::new("cat"), &[], "prompt text", 30).unwrap(); assert_eq!(out, "prompt text"); } #[test] fn run_cli_kills_a_child_that_overruns_its_timeout() { let err = run_cli(Path::new("sleep"), &["30"], "", 1).unwrap_err().to_string(); assert!(err.contains("timed out"), "got: {err}"); } #[test] fn run_cli_reports_a_nonzero_exit() { let err = run_cli(Path::new("false"), &[], "", 30).unwrap_err().to_string(); assert!(err.contains("exited with"), "got: {err}"); } }