1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-10-09 21:03:17 +02:00

fix: bound codex fallback and record resolved model

This commit is contained in:
archivr-qa 2026-08-24 15:31:10 +02:00
parent b4f6b9fc69
commit d56fd8b41b
No known key found for this signature in database
5 changed files with 466 additions and 167 deletions

View file

@ -1,6 +1,6 @@
use anyhow::{Context, Result, bail};
use anyhow::{bail, Context, Result};
use chrono::Utc;
use rusqlite::{Connection, OptionalExtension, params};
use rusqlite::{params, Connection, OptionalExtension};
use std::path::{Path, PathBuf};
use uuid::Uuid;
@ -121,6 +121,8 @@ pub struct EntrySummaryRecord {
pub summary_uid: String,
pub entry_uid: String,
pub provider_kind: String,
/// Model selected by the provider after resolving the requested cache-key alias.
pub resolved_model: Option<String>,
pub provider_model: Option<String>,
pub prompt_version: String,
pub input_sha256: String,
@ -350,6 +352,7 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
entry_id INTEGER NOT NULL REFERENCES archived_entries(id) ON DELETE CASCADE,
provider_kind TEXT NOT NULL,
provider_model TEXT,
resolved_model TEXT,
prompt_version TEXT NOT NULL,
input_sha256 TEXT NOT NULL,
status TEXT NOT NULL CHECK(status IN ('pending','running','completed','failed')),
@ -476,6 +479,12 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
// Migration: add notes_json column to existing capture_jobs tables.
// Silently ignored when the column already exists (idempotent).
let _ = conn.execute("ALTER TABLE capture_jobs ADD COLUMN notes_json TEXT", []);
// Provider responses may resolve a requested alias to a concrete model.
// Keep that display-only value outside the cache key.
let _ = conn.execute(
"ALTER TABLE entry_summaries ADD COLUMN resolved_model TEXT",
[],
);
// Migration: add requires_auth column to existing collections tables.
// Silently ignored when the column already exists (idempotent).
let _ = conn.execute(
@ -1403,7 +1412,8 @@ pub fn fail_stalled_capture_jobs(conn: &Connection) -> Result<usize> {
/// `SELECT` list shared by every `entry_summaries` read, so all readers build
/// an identical `EntrySummaryRecord` from the same column ordering.
const ENTRY_SUMMARY_COLS: &str = "SELECT s.summary_uid, e.entry_uid, s.provider_kind, s.provider_model,
const ENTRY_SUMMARY_COLS: &str =
"SELECT s.summary_uid, e.entry_uid, s.provider_kind, s.provider_model, s.resolved_model,
s.prompt_version, s.input_sha256, s.status, s.summary_text,
s.error_text, s.created_at, s.updated_at, s.completed_at
FROM entry_summaries s
@ -1416,17 +1426,16 @@ fn map_entry_summary(row: &rusqlite::Row<'_>) -> rusqlite::Result<EntrySummaryRe
provider_kind: row.get(2)?,
// '' is the stored stand-in for "this provider has no model"; see the
// doc comment on EntrySummaryRecord for why it is not NULL.
provider_model: row
.get::<_, Option<String>>(3)?
.filter(|m| !m.is_empty()),
prompt_version: row.get(4)?,
input_sha256: row.get(5)?,
status: row.get(6)?,
summary_text: row.get(7)?,
error_text: row.get(8)?,
created_at: row.get(9)?,
updated_at: row.get(10)?,
completed_at: row.get(11)?,
provider_model: row.get::<_, Option<String>>(3)?.filter(|m| !m.is_empty()),
resolved_model: row.get(4)?,
prompt_version: row.get(5)?,
input_sha256: row.get(6)?,
status: row.get(7)?,
summary_text: row.get(8)?,
error_text: row.get(9)?,
created_at: row.get(10)?,
updated_at: row.get(11)?,
completed_at: row.get(12)?,
})
}
@ -1464,7 +1473,7 @@ pub fn upsert_pending_entry_summary(
input_sha256, status, summary_text, error_text, created_at, updated_at, completed_at)
VALUES (?1, ?2, ?3, ?4, ?5, ?6, 'pending', NULL, NULL, ?7, ?7, NULL)
ON CONFLICT(entry_id, provider_kind, provider_model, prompt_version, input_sha256)
DO UPDATE SET status = 'pending', summary_text = NULL, error_text = NULL,
DO UPDATE SET status = 'pending', summary_text = NULL, error_text = NULL, resolved_model = NULL,
completed_at = NULL, updated_at = ?7",
rusqlite::params![
summary_uid,
@ -1488,6 +1497,25 @@ pub fn upsert_pending_entry_summary(
Ok(stored)
}
/// Completes a summary while retaining the provider's concrete response model
/// for display. The requested provider model remains the cache-key identity.
pub fn update_entry_summary_completed(
conn: &Connection,
summary_uid: &str,
summary_text: &str,
resolved_model: Option<&str>,
) -> Result<()> {
let now = now_timestamp();
conn.execute(
"UPDATE entry_summaries
SET status = 'completed', summary_text = ?1, error_text = NULL,
resolved_model = ?2, completed_at = ?3, updated_at = ?3
WHERE summary_uid = ?4",
rusqlite::params![summary_text, resolved_model, now, summary_uid],
)?;
Ok(())
}
/// Moves a summary row through `running` → `completed` / `failed`.
/// `completed_at` is stamped only on the terminal `completed` transition.
pub fn update_entry_summary_status(
@ -1508,7 +1536,14 @@ pub fn update_entry_summary_status(
SET status = ?1, summary_text = ?2, error_text = ?3,
completed_at = ?4, updated_at = ?5
WHERE summary_uid = ?6",
rusqlite::params![status, summary_text, error_text, completed_at, now, summary_uid],
rusqlite::params![
status,
summary_text,
error_text,
completed_at,
now,
summary_uid
],
)?;
Ok(())
}
@ -1665,7 +1700,11 @@ pub fn finish_archive_run(conn: &Connection, run_id: i64) -> Result<()> {
[run_id],
|row| row.get(0),
)?;
let status = if failed_count > 0 { "failed" } else { "completed" };
let status = if failed_count > 0 {
"failed"
} else {
"completed"
};
conn.execute(
"UPDATE archive_runs SET status = ?1, finished_at = ?2 WHERE id = ?3",
params![status, now_timestamp(), run_id],
@ -1901,9 +1940,8 @@ pub fn delete_entry(conn: &Connection, entry_uid: &str) -> Result<bool> {
// (no grandchildren), so without `id = ?1` the set would be empty and
// cascade_cached_bytes_after_subtree_delete would not recalculate shared-blob totals.
let subtree_ids: Vec<i64> = {
let mut stmt = conn.prepare(
"SELECT id FROM archived_entries WHERE id = ?1 OR root_entry_id = ?1",
)?;
let mut stmt =
conn.prepare("SELECT id FROM archived_entries WHERE id = ?1 OR root_entry_id = ?1")?;
stmt.query_map([entry_id], |row| row.get(0))?
.collect::<rusqlite::Result<_>>()?
};
@ -2554,7 +2592,6 @@ pub fn visibility_to_bits(visibility: &str) -> u32 {
}
}
/// Returns the id of the '_default_' collection, creating it if absent.
pub fn ensure_default_collection(conn: &Connection) -> Result<i64> {
let now = now_timestamp();
@ -2725,7 +2762,12 @@ pub fn get_entry_collection_memberships(
WHERE ce.entry_id = ?1",
)?;
stmt.query_map([entry_id], |row| {
Ok((row.get(0)?, row.get(1)?, row.get(2)?, row.get::<_, i64>(3)? as u32))
Ok((
row.get(0)?,
row.get(1)?,
row.get(2)?,
row.get::<_, i64>(3)? as u32,
))
})?
.collect::<Result<_, _>>()
.map_err(Into::into)
@ -3283,7 +3325,10 @@ mod tests {
|row| row.get::<_, i64>(0).map(|v| v as u32),
)
.unwrap();
assert_eq!(default_bits, 2, "default collection should start USER-visible");
assert_eq!(
default_bits, 2,
"default collection should start USER-visible"
);
// Create an entry with visibility = "private" (what capture always passes).
let entry = create_entry_fixture(&conn, "private", None, None);
@ -3331,7 +3376,10 @@ mod tests {
|row| row.get::<_, i64>(0).map(|v| v as u32),
)
.unwrap();
assert_eq!(public_bits, 3, "collection default=public should produce bits=3");
assert_eq!(
public_bits, 3,
"collection default=public should produce bits=3"
);
// Child entries must NOT use the collection default — they keep
// visibility_to_bits(entry.visibility) so parent-child visibility
@ -3344,7 +3392,10 @@ mod tests {
|row| row.get::<_, i64>(0).map(|v| v as u32),
)
.unwrap();
assert_eq!(child_bits, 0, "child entries must use visibility_to_bits, not collection default");
assert_eq!(
child_bits, 0,
"child entries must use visibility_to_bits, not collection default"
);
}
#[test]
@ -3680,11 +3731,9 @@ mod tests {
assert_eq!(cs_new.full_path, "/natural-science/cs");
// /science/cs/algorithms must have moved
assert!(
get_tag_by_path(&conn, "/science/cs/algorithms")
.unwrap()
.is_none()
);
assert!(get_tag_by_path(&conn, "/science/cs/algorithms")
.unwrap()
.is_none());
let algo_new = get_tag_by_uid(&conn, &algo.tag_uid).unwrap().unwrap();
assert_eq!(algo_new.full_path, "/natural-science/cs/algorithms");
}
@ -4373,30 +4422,50 @@ mod tests {
// Root item (parent_item_id IS NULL) — mirrors what record_container_entry does.
let root_item = create_archive_run_item(
&c, run.id, None, 0, "https://example.com/pl", None, "youtube", "playlist",
).unwrap();
&c,
run.id,
None,
0,
"https://example.com/pl",
None,
"youtube",
"playlist",
)
.unwrap();
// Child item (parent_item_id IS NOT NULL).
let child_item = create_archive_run_item(
&c, run.id, Some(root_item.id), 1, "https://example.com/v1", None, "youtube", "video",
).unwrap();
&c,
run.id,
Some(root_item.id),
1,
"https://example.com/v1",
None,
"youtube",
"video",
)
.unwrap();
// Complete both — marks archive_runs.completed_count = 2.
c.execute(
"UPDATE archive_run_items SET status = 'completed' WHERE id IN (?1, ?2)",
rusqlite::params![root_item.id, child_item.id],
).unwrap();
)
.unwrap();
refresh_run_counters(&c, run.id).unwrap();
let total: i64 = c.query_row(
"SELECT completed_count FROM archive_runs WHERE id = ?1", [run.id], |r| r.get(0),
).unwrap();
let total: i64 = c
.query_row(
"SELECT completed_count FROM archive_runs WHERE id = ?1",
[run.id],
|r| r.get(0),
)
.unwrap();
assert_eq!(total, 2, "both items completed: DB counter must be 2");
let child_count = get_run_completed_child_count(&c, run.id).unwrap();
assert_eq!(child_count, 1, "only the child item must be counted");
}
// ── entry_summaries ────────────────────────────────────────────────────
#[test]
@ -4421,6 +4490,41 @@ mod tests {
assert!(latest_entry_summary(&c, entry.id).unwrap().is_some());
}
#[test]
fn initialize_schema_migrates_resolved_model_for_existing_summary_table() {
let c = Connection::open_in_memory().unwrap();
c.execute_batch(
"CREATE TABLE entry_summaries (
id INTEGER PRIMARY KEY,
summary_uid TEXT NOT NULL UNIQUE,
entry_id INTEGER NOT NULL,
provider_kind TEXT NOT NULL,
provider_model TEXT,
prompt_version TEXT NOT NULL,
input_sha256 TEXT NOT NULL,
status TEXT NOT NULL,
summary_text TEXT,
error_text TEXT,
created_at TEXT NOT NULL,
updated_at TEXT NOT NULL,
completed_at TEXT,
UNIQUE(entry_id, provider_kind, provider_model, prompt_version, input_sha256)
);",
)
.unwrap();
initialize_schema(&c).unwrap();
let has_resolved_model: i64 = c
.query_row(
"SELECT COUNT(*) FROM pragma_table_info('entry_summaries') WHERE name = 'resolved_model'",
[],
|r| r.get(0),
)
.unwrap();
assert_eq!(has_resolved_model, 1);
}
#[test]
fn entry_summary_lifecycle_pending_running_completed() {
let c = conn();
@ -4451,6 +4555,48 @@ mod tests {
assert!(rec.completed_at.is_some(), "completed rows must be stamped");
}
#[test]
fn entry_summary_keeps_requested_model_for_cache_and_resolved_model_for_display() {
let c = conn();
let entry = create_entry_fixture(&c, "private", None, None);
let uid = upsert_pending_entry_summary(
&c,
entry.id,
"anthropic_http",
Some("claude-3-5-sonnet-latest"),
"v1",
"sha1",
)
.unwrap();
update_entry_summary_completed(&c, &uid, "summary", Some("claude-3-5-sonnet-20241022"))
.unwrap();
let rec = get_entry_summary_by_uid(&c, &uid).unwrap().unwrap();
assert_eq!(
rec.provider_model.as_deref(),
Some("claude-3-5-sonnet-latest")
);
assert_eq!(
rec.resolved_model.as_deref(),
Some("claude-3-5-sonnet-20241022")
);
assert_eq!(
find_entry_summary(
&c,
entry.id,
"anthropic_http",
Some("claude-3-5-sonnet-latest"),
"v1",
"sha1",
)
.unwrap()
.unwrap()
.summary_uid,
uid,
);
}
#[test]
fn entry_summary_failure_records_error_and_no_completed_at() {
let c = conn();
@ -4474,7 +4620,10 @@ mod tests {
// Regenerating the same key must reset the row in place, not add a second.
let second =
upsert_pending_entry_summary(&c, entry.id, "claude_cli", None, "v1", "sha").unwrap();
assert_eq!(first, second, "same cache key must keep the same summary_uid");
assert_eq!(
first, second,
"same cache key must keep the same summary_uid"
);
let n: i64 = c
.query_row("SELECT COUNT(*) FROM entry_summaries", [], |r| r.get(0))
@ -4530,9 +4679,16 @@ mod tests {
.unwrap();
assert_eq!(hit.summary_uid, uid);
assert!(
find_entry_summary(&c, entry.id, "openai_compatible", Some("gpt"), "v1", "changed")
.unwrap()
.is_none(),
find_entry_summary(
&c,
entry.id,
"openai_compatible",
Some("gpt"),
"v1",
"changed"
)
.unwrap()
.is_none(),
"a changed input digest must miss the cache"
);
}
@ -4546,7 +4702,13 @@ mod tests {
update_entry_summary_status(&c, &a, "completed", Some("first"), None).unwrap();
update_entry_summary_status(&c, &b, "completed", Some("second"), None).unwrap();
// Same-timestamp ties break on id DESC, so the later insert wins either way.
assert_eq!(latest_entry_summary(&c, entry.id).unwrap().unwrap().summary_uid, b);
assert_eq!(
latest_entry_summary(&c, entry.id)
.unwrap()
.unwrap()
.summary_uid,
b
);
}
#[test]
@ -4573,7 +4735,10 @@ mod tests {
fn entry_id_for_uid_resolves_and_misses() {
let c = conn();
let entry = create_entry_fixture(&c, "private", None, None);
assert_eq!(entry_id_for_uid(&c, &entry.entry_uid).unwrap(), Some(entry.id));
assert_eq!(
entry_id_for_uid(&c, &entry.entry_uid).unwrap(),
Some(entry.id)
);
assert_eq!(entry_id_for_uid(&c, "ent_nope").unwrap(), None);
}
}

View file

@ -13,7 +13,7 @@
//! `ARCHIVR_SINGLE_FILE`, `ARCHIVR_TWEET_SCRAPER`, …) and keeping API keys out
//! of any file the archive would otherwise persist.
use anyhow::{Context, Result, anyhow, bail};
use anyhow::{anyhow, bail, Context, Result};
use base64::Engine;
use std::{
env,
@ -253,13 +253,19 @@ fn resolve_cli(env_name: &str, well_known_absolute: &[&str], bare: &str) -> Path
pub fn provider_from_env(kind: &str) -> Result<ProviderConfig> {
match kind {
"anthropic_http" => Ok(ProviderConfig::AnthropicHttp(HttpProviderConfig {
endpoint: env_or("ARCHIVR_ANTHROPIC_URL", "https://api.anthropic.com/v1/messages"),
endpoint: env_or(
"ARCHIVR_ANTHROPIC_URL",
"https://api.anthropic.com/v1/messages",
),
api_key: required_env("ARCHIVR_ANTHROPIC_API_KEY")?,
model: env_or("ARCHIVR_ANTHROPIC_MODEL", "claude-3-5-sonnet-latest"),
timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS),
})),
"openai_compatible" => Ok(ProviderConfig::OpenAiCompatible(HttpProviderConfig {
endpoint: env_or("ARCHIVR_OPENAI_URL", "https://api.openai.com/v1/chat/completions"),
endpoint: env_or(
"ARCHIVR_OPENAI_URL",
"https://api.openai.com/v1/chat/completions",
),
api_key: required_env("ARCHIVR_OPENAI_API_KEY")?,
model: env_or("ARCHIVR_OPENAI_MODEL", "gpt-4o-mini"),
timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS),
@ -328,8 +334,12 @@ fn http_client(timeout_secs: u64) -> Result<reqwest::blocking::Client> {
/// Body builder kept separate from the transport so it can be unit-tested
/// without a network round-trip.
fn read_image_base64(image: &SummaryImage) -> Result<String> {
let bytes = std::fs::read(&image.archive_file)
.with_context(|| format!("failed to read summary image {}", image.archive_file.display()))?;
let bytes = std::fs::read(&image.archive_file).with_context(|| {
format!(
"failed to read summary image {}",
image.archive_file.display()
)
})?;
Ok(base64::engine::general_purpose::STANDARD.encode(bytes))
}
@ -420,7 +430,10 @@ impl SummaryProvider for AnthropicHttpProvider {
let status = resp.status();
let text = resp.text().unwrap_or_default();
if !status.is_success() {
bail!("anthropic API returned {status}: {}", truncate_for_error(&text));
bail!(
"anthropic API returned {status}: {}",
truncate_for_error(&text)
);
}
parse_anthropic_response(&text)
}
@ -499,12 +512,7 @@ fn truncate_for_error(s: &str) -> String {
/// the child if it overruns. stdin is written on a third thread because a
/// 48 KB prompt can exceed the pipe buffer, and writing it inline would
/// deadlock against a child that is waiting for us to read its output.
fn run_cli(
executable: &Path,
args: &[&str],
prompt: &str,
timeout_secs: u64,
) -> Result<String> {
fn run_cli(executable: &Path, args: &[&str], prompt: &str, timeout_secs: u64) -> Result<String> {
let mut child = Command::new(executable)
.args(args)
.stdin(Stdio::piped())
@ -538,14 +546,13 @@ fn run_cli(
});
let collected = match rx.recv_timeout(Duration::from_secs(timeout_secs)) {
Ok(res) => res.with_context(|| format!("failed to read stdout of {}", executable.display()))?,
Ok(res) => {
res.with_context(|| format!("failed to read stdout of {}", executable.display()))?
}
Err(_) => {
let _ = child.kill();
let _ = child.wait();
bail!(
"{} timed out after {timeout_secs}s",
executable.display()
);
bail!("{} timed out after {timeout_secs}s", executable.display());
}
};
@ -588,7 +595,12 @@ impl SummaryProvider for ClaudeCliProvider {
args.push("--model");
args.push(model);
}
let out = run_cli(&self.0.executable, &args, &build_combined_prompt(request), self.0.timeout_secs)?;
let out = run_cli(
&self.0.executable,
&args,
&build_combined_prompt(request),
self.0.timeout_secs,
)?;
Ok(SummaryOutput {
text: out,
model: self.0.model.clone(),
@ -638,7 +650,11 @@ mod codex {
let out = std::fs::read_to_string(path).ok()?;
let _ = std::fs::remove_file(path);
let trimmed = out.trim();
if trimmed.is_empty() { None } else { Some(trimmed.to_string()) }
if trimmed.is_empty() {
None
} else {
Some(trimmed.to_string())
}
}
pub fn primary_args(
@ -698,30 +714,21 @@ mod codex {
// Fallback: positional prompt, no stdin, same --output-last-message.
let fb = positional_args(images, &out_path, cfg.model.as_deref(), prompt);
let out = Command::new(&cfg.executable)
.args(&fb)
.output()
.with_context(|| {
format!(
"codex `exec -` failed ({primary_err:#}); positional fallback also failed to spawn{}",
missing_binary_hint(cfg)
)
})?;
if !out.status.success() {
let fb_refs: Vec<&str> = fb.iter().map(String::as_str).collect();
let out = run_cli(&cfg.executable, &fb_refs, "", cfg.timeout_secs).with_context(|| {
let _ = std::fs::remove_file(&out_path);
bail!(
"codex `exec -` failed ({primary_err:#}); positional fallback exited with {}: {}",
out.status,
truncate_for_error(&String::from_utf8_lossy(&out.stderr))
);
}
format!(
"codex `exec -` failed ({primary_err:#}); positional fallback failed{}",
missing_binary_hint(cfg)
)
})?;
if let Some(text) = read_and_cleanup(&out_path) {
return Ok(text);
}
// Last resort — the child succeeded but wrote nothing to the file. Fall
// back to raw stdout so the caller has *something* to normalize.
let _ = std::fs::remove_file(&out_path);
Ok(String::from_utf8_lossy(&out.stdout).to_string())
Ok(out)
}
}
@ -771,7 +778,8 @@ pub fn strip_html(html: &str) -> String {
.unwrap();
let comments = regex::Regex::new(r"(?s)<!--.*?-->").unwrap();
// Treat block-level closers as line breaks so paragraphs do not run together.
let breaks = regex::Regex::new(r"(?i)</\s*(p|div|br|li|h[1-6]|tr|section|article)\s*>").unwrap();
let breaks =
regex::Regex::new(r"(?i)</\s*(p|div|br|li|h[1-6]|tr|section|article)\s*>").unwrap();
let tags = regex::Regex::new(r"(?s)<[^>]*>").unwrap();
let spaces = regex::Regex::new(r"[ \t\r\f\v]+").unwrap();
let blank_lines = regex::Regex::new(r"\n{3,}").unwrap();
@ -782,11 +790,7 @@ pub fn strip_html(html: &str) -> String {
let s = tags.replace_all(&s, " ");
let s = decode_entities(&s);
let s = spaces.replace_all(&s, " ");
let s = s
.lines()
.map(str::trim)
.collect::<Vec<_>>()
.join("\n");
let s = s.lines().map(str::trim).collect::<Vec<_>>().join("\n");
blank_lines.replace_all(&s, "\n\n").trim().to_string()
}
@ -948,9 +952,12 @@ fn load_summary_image_candidates(
store_path: &Path,
entry_id: i64,
) -> Result<Vec<SummaryImage>> {
let canonical_store = store_path
.canonicalize()
.with_context(|| format!("failed to canonicalize store path: {}", store_path.display()))?;
let canonical_store = store_path.canonicalize().with_context(|| {
format!(
"failed to canonicalize store path: {}",
store_path.display()
)
})?;
let mut stmt = conn.prepare(
"SELECT b.sha256, b.mime_type, b.extension, b.byte_size, ea.relpath
FROM entry_artifacts ea
@ -1050,7 +1057,11 @@ pub fn build_summary_input(
// Load every matching artifact in insertion order so a thread summarizes
// as the whole conversation, not just its first status.
let is_tweetish = matches!(entity_kind.as_str(), "tweet" | "tweet_thread");
let primary_role = if is_tweetish { "raw_tweet_json" } else { "primary_media" };
let primary_role = if is_tweetish {
"raw_tweet_json"
} else {
"primary_media"
};
let mut artifacts = load_summary_artifacts(&conn, entry_id, primary_role)?;
if artifacts.is_empty() && is_tweetish {
// Older archives may have stored tweet payloads under `primary_media`.
@ -1066,8 +1077,11 @@ pub fn build_summary_input(
let ext = extension_of(relpath);
let mime = mime_opt.clone().unwrap_or_default();
let piece = if ext == "md" || ext == "markdown" || ext == "txt"
|| mime.starts_with("text/markdown") || mime == "text/plain"
let piece = if ext == "md"
|| ext == "markdown"
|| ext == "txt"
|| mime.starts_with("text/markdown")
|| mime == "text/plain"
{
std::fs::read_to_string(&abs)
.with_context(|| format!("failed to read {}", abs.display()))?
@ -1189,23 +1203,16 @@ pub fn summarize_entry(
match provider.summarize(&input.request) {
Ok(output) => {
let text = normalize_summary_json(&output.text);
database::update_entry_summary_status(
database::update_entry_summary_completed(
&conn,
&summary_uid,
"completed",
Some(&text),
None,
&text,
output.model.as_deref(),
)?;
}
Err(e) => {
let msg = format!("{e:#}");
database::update_entry_summary_status(
&conn,
&summary_uid,
"failed",
None,
Some(&msg),
)?;
database::update_entry_summary_status(&conn, &summary_uid, "failed", None, Some(&msg))?;
return Err(e);
}
}
@ -1297,7 +1304,9 @@ mod tests {
fn provider_from_env_openai_missing_key_names_the_variable() {
let _g = ENV_LOCK.lock().unwrap();
clear_provider_env();
let err = provider_from_env("openai_compatible").unwrap_err().to_string();
let err = provider_from_env("openai_compatible")
.unwrap_err()
.to_string();
assert!(err.contains("ARCHIVR_OPENAI_API_KEY"), "got: {err}");
}
@ -1307,7 +1316,10 @@ mod tests {
clear_provider_env();
unsafe {
env::set_var("ARCHIVR_OPENAI_API_KEY", "k");
env::set_var("ARCHIVR_OPENAI_URL", "http://localhost:1234/v1/chat/completions");
env::set_var(
"ARCHIVR_OPENAI_URL",
"http://localhost:1234/v1/chat/completions",
);
env::set_var("ARCHIVR_OPENAI_MODEL", "local-model");
env::set_var("ARCHIVR_SUMMARY_HTTP_TIMEOUT", "7");
}
@ -1333,7 +1345,10 @@ mod tests {
// file-name-`claude` path.
assert!(
c.executable == PathBuf::from("claude")
|| c.executable.file_name().map(|f| f == "claude").unwrap_or(false),
|| c.executable
.file_name()
.map(|f| f == "claude")
.unwrap_or(false),
"unexpected claude executable: {}",
c.executable.display()
);
@ -1348,7 +1363,10 @@ mod tests {
// deliberately opportunistic.
assert!(
c.executable == PathBuf::from("codex")
|| c.executable.file_name().map(|f| f == "codex").unwrap_or(false),
|| c.executable
.file_name()
.map(|f| f == "codex")
.unwrap_or(false),
"unexpected codex executable: {}",
c.executable.display()
);
@ -1409,18 +1427,14 @@ mod tests {
assert_eq!(body["model"], "gpt-4o-mini");
assert_eq!(body["messages"][0]["role"], "system");
assert_eq!(body["messages"][1]["role"], "user");
assert!(
body["messages"][1]["content"]
.as_str()
.unwrap()
.contains("Body text.")
);
assert!(
!body["messages"][1]["content"]
.as_str()
.unwrap()
.contains("\"tldr\"")
);
assert!(body["messages"][1]["content"]
.as_str()
.unwrap()
.contains("Body text."));
assert!(!body["messages"][1]["content"]
.as_str()
.unwrap()
.contains("\"tldr\""));
}
#[test]
@ -1459,29 +1473,57 @@ mod tests {
.position(|arg| arg == "--output-last-message")
.unwrap();
assert!(image_at < output_at);
assert_eq!(args[image_at + 1], request.images[0].archive_file.to_string_lossy());
assert_eq!(
args[image_at + 1],
request.images[0].archive_file.to_string_lossy()
);
assert_eq!(args.last().unwrap(), "-");
}
#[test]
fn codex_positional_arguments_put_images_before_output_path() {
let (_temp, request) = image_request();
let args = codex::positional_args(
&request.images,
Path::new("/tmp/output"),
None,
"prompt",
);
let args =
codex::positional_args(&request.images, Path::new("/tmp/output"), None, "prompt");
let image_at = args.iter().position(|arg| arg == "--image").unwrap();
let output_at = args
.iter()
.position(|arg| arg == "--output-last-message")
.unwrap();
assert!(image_at < output_at);
assert_eq!(args[image_at + 1], request.images[0].archive_file.to_string_lossy());
assert_eq!(
args[image_at + 1],
request.images[0].archive_file.to_string_lossy()
);
assert_eq!(args.last().unwrap(), "prompt");
}
#[cfg(unix)]
#[test]
fn codex_positional_fallback_honors_cli_timeout() {
use std::os::unix::fs::PermissionsExt;
let temp = tempfile::tempdir().unwrap();
let executable = temp.path().join("codex-fixture.sh");
std::fs::write(
&executable,
"#!/bin/sh\nfor arg in \"$@\"; do [ \"$arg\" = \"-\" ] && exit 1; done\nsleep 30\n",
)
.unwrap();
std::fs::set_permissions(&executable, std::fs::Permissions::from_mode(0o755)).unwrap();
let cfg = CliProviderConfig {
executable,
model: None,
timeout_secs: 1,
};
let started = std::time::Instant::now();
let err = format!("{:#}", codex::run(&cfg, "prompt", &[]).unwrap_err());
assert!(started.elapsed() < Duration::from_secs(5));
assert!(err.contains("timed out after 1s"), "got: {err}");
}
#[test]
fn claude_rejects_images_before_spawning() {
let (_temp, request) = image_request();
@ -1491,12 +1533,16 @@ mod tests {
timeout_secs: 1,
});
let err = provider.summarize(&request).unwrap_err().to_string();
assert!(err.contains("Claude CLI cannot attach local images"), "got: {err}");
assert!(
err.contains("Claude CLI cannot attach local images"),
"got: {err}"
);
}
#[test]
fn parse_anthropic_response_extracts_text_and_model() {
let body = r#"{"model":"claude-3-5-sonnet-20241022","content":[{"type":"text","text":"hi"}]}"#;
let body =
r#"{"model":"claude-3-5-sonnet-20241022","content":[{"type":"text","text":"hi"}]}"#;
let out = parse_anthropic_response(body).unwrap();
assert_eq!(out.text, "hi");
assert_eq!(out.model.as_deref(), Some("claude-3-5-sonnet-20241022"));
@ -1539,7 +1585,10 @@ mod tests {
#[test]
fn strip_html_decodes_common_entities() {
assert_eq!(strip_html("<p>a &amp; b &nbsp;c</p>").replace('\u{a0}', " "), "a & b c");
assert_eq!(
strip_html("<p>a &amp; b &nbsp;c</p>").replace('\u{a0}', " "),
"a & b c"
);
}
#[test]
@ -1707,8 +1756,12 @@ mod tests {
.unwrap();
}
let input = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()).unwrap();
assert!(input.request.content.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body."));
let input =
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()).unwrap();
assert!(input
.request
.content
.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body."));
}
fn summary_image_fixture() -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
@ -1780,19 +1833,12 @@ mod tests {
)
.unwrap();
let no_artifact = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err();
let no_artifact =
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err();
assert!(is_unsupported_summary_content_error(&no_artifact));
add_summary_image_artifact(
&paths,
entry.id,
99,
"primary_media",
"mp4",
"video/mp4",
1,
);
add_summary_image_artifact(&paths, entry.id, 99, "primary_media", "mp4", "video/mp4", 1);
let video = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err();
assert!(is_unsupported_summary_content_error(&video));
@ -1819,18 +1865,25 @@ mod tests {
)
.unwrap();
drop(conn);
let empty_text = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err();
let empty_text =
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err();
assert!(is_unsupported_summary_content_error(&empty_text));
std::fs::remove_file(paths.store_path.join(empty_relpath)).unwrap();
let read_error = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err();
let read_error =
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err();
assert!(!is_unsupported_summary_content_error(&read_error));
assert_eq!(UNSUPPORTED_SUMMARY_CONTENT_MESSAGE, "This entry can’t be summarized yet.\n\nIt doesn’t contain archived text that a summary provider can read. Summaries currently support text notes, web pages, X posts and threads, and X Articles. Video, audio, and image-only entries need a transcript or text source.");
assert!(!is_unsupported_summary_content_error(&anyhow!("provider timeout")));
assert!(!is_unsupported_summary_content_error(&anyhow!("entry not found: {}", entry.entry_uid)));
assert!(!is_unsupported_summary_content_error(&anyhow!(
"provider timeout"
)));
assert!(!is_unsupported_summary_content_error(&anyhow!(
"entry not found: {}",
entry.entry_uid
)));
}
fn add_summary_image_artifact(
@ -1897,13 +1950,17 @@ mod tests {
let text_only = build_summary_input(
&paths,
&entry.entry_uid,
SummaryBuildOptions { include_images: false },
SummaryBuildOptions {
include_images: false,
},
)
.unwrap();
let visual = build_summary_input(
&paths,
&entry.entry_uid,
SummaryBuildOptions { include_images: true },
SummaryBuildOptions {
include_images: true,
},
)
.unwrap();
@ -1934,21 +1991,52 @@ mod tests {
#[test]
fn summary_image_selection_enforces_aggregate_limit_and_keeps_scanning() {
let (_temp, paths, entry) = summary_image_fixture();
add_summary_image_artifact(&paths, entry.id, 0, "media", "jpg", "image/jpeg", 4 * 1024 * 1024);
add_summary_image_artifact(&paths, entry.id, 1, "media", "png", "image/png", 4 * 1024 * 1024);
add_summary_image_artifact(&paths, entry.id, 2, "media", "webp", "image/webp", 4 * 1024 * 1024);
add_summary_image_artifact(
&paths,
entry.id,
0,
"media",
"jpg",
"image/jpeg",
4 * 1024 * 1024,
);
add_summary_image_artifact(
&paths,
entry.id,
1,
"media",
"png",
"image/png",
4 * 1024 * 1024,
);
add_summary_image_artifact(
&paths,
entry.id,
2,
"media",
"webp",
"image/webp",
4 * 1024 * 1024,
);
add_summary_image_artifact(&paths, entry.id, 3, "media", "gif", "image/gif", 1);
let visual = build_summary_input(
&paths,
&entry.entry_uid,
SummaryBuildOptions { include_images: true },
SummaryBuildOptions {
include_images: true,
},
)
.unwrap();
assert_eq!(visual.request.images.len(), 3);
assert_eq!(
visual.request.images.iter().map(|image| image.byte_size).sum::<u64>(),
visual
.request
.images
.iter()
.map(|image| image.byte_size)
.sum::<u64>(),
MAX_SUMMARY_IMAGE_TOTAL_BYTES
);
assert!(visual
@ -1958,6 +2046,45 @@ mod tests {
.all(|image| image.sha256 != "summary-image-03"));
}
#[test]
fn summarize_entry_uses_requested_alias_for_cache_and_response_model_for_display() {
struct ResolvedModelProvider;
impl SummaryProvider for ResolvedModelProvider {
fn kind(&self) -> &'static str {
"anthropic_http"
}
fn model(&self) -> Option<&str> {
Some("claude-3-5-sonnet-latest")
}
fn summarize(&self, _: &SummaryRequest) -> Result<SummaryOutput> {
Ok(SummaryOutput {
text: r#"{"tldr":"t","summary":"s","tags":[]}"#.to_string(),
model: Some("claude-3-5-sonnet-20241022".to_string()),
})
}
}
let (_temp, paths, entry) = summary_image_fixture();
let record = summarize_entry(
&paths,
&entry.entry_uid,
SummaryBuildOptions::default(),
&ResolvedModelProvider,
"v1",
)
.unwrap();
assert_eq!(
record.provider_model.as_deref(),
Some("claude-3-5-sonnet-latest")
);
assert_eq!(
record.resolved_model.as_deref(),
Some("claude-3-5-sonnet-20241022")
);
}
// ── Output normalization ───────────────────────────────────────────────
#[test]
@ -1977,7 +2104,8 @@ mod tests {
#[test]
fn normalize_summary_json_recovers_json_wrapped_in_prose() {
let raw = "Sure! Here you go:\n{\"tldr\":\"t\",\"summary\":\"s\",\"tags\":[]}\nHope that helps.";
let raw =
"Sure! Here you go:\n{\"tldr\":\"t\",\"summary\":\"s\",\"tags\":[]}\nHope that helps.";
let v: serde_json::Value = serde_json::from_str(&normalize_summary_json(raw)).unwrap();
assert_eq!(v["tldr"], "t");
}
@ -2004,13 +2132,17 @@ mod tests {
#[test]
fn run_cli_kills_a_child_that_overruns_its_timeout() {
let err = run_cli(Path::new("sleep"), &["30"], "", 1).unwrap_err().to_string();
let err = run_cli(Path::new("sleep"), &["30"], "", 1)
.unwrap_err()
.to_string();
assert!(err.contains("timed out"), "got: {err}");
}
#[test]
fn run_cli_reports_a_nonzero_exit() {
let err = run_cli(Path::new("false"), &[], "", 30).unwrap_err().to_string();
let err = run_cli(Path::new("false"), &[], "", 30)
.unwrap_err()
.to_string();
assert!(err.contains("exited with"), "got: {err}");
}
}

View file

@ -6,7 +6,7 @@
<title>Archivr</title>
<link rel="icon" type="image/svg+xml" href="/favicon.svg">
<link rel="icon" type="image/x-icon" href="/favicon.ico">
<script type="module" crossorigin src="/assets/index-DH0j8cUl.js"></script>
<script type="module" crossorigin src="/assets/index-DMKGhxnA.js"></script>
<link rel="stylesheet" crossorigin href="/assets/index-1h0SqIvL.css">
</head>
<body>