1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-10-09 12:55:00 +02:00

Merge branch 'sol-summary-core' into integration-all-three

# Conflicts:
#	crates/archivr-core/src/database.rs
#	crates/archivr-core/src/summarizer.rs
This commit is contained in:
archivr-qa 2026-08-24 16:29:49 +02:00
commit 26df7dd2e9
No known key found for this signature in database
6 changed files with 597 additions and 218 deletions

View file

@ -951,7 +951,10 @@ fn register_tweet_artifacts(
})?; })?;
for (role, raw_relpath) in tweet_raw_artifacts(&json_str)? { for (role, raw_relpath) in tweet_raw_artifacts(&json_str)? {
let raw_path = PathBuf::from(&raw_relpath); let raw_path = PathBuf::from(&raw_relpath);
let blob = blob_record_for_raw_relpath(store_path, &raw_path)?; let mut blob = blob_record_for_raw_relpath(store_path, &raw_path)?;
if role == "media" {
blob.mime_type = tweet_media_image_mime(blob.extension.as_deref());
}
let blob_id = database::upsert_blob(conn, &blob)?; let blob_id = database::upsert_blob(conn, &blob)?;
database::add_entry_artifact( database::add_entry_artifact(
conn, conn,
@ -1030,6 +1033,22 @@ fn record_tweet_entry(
Ok(entry) Ok(entry)
} }
/// Trusted image MIME types emitted by the X downloader's media paths.
///
/// Tweet JSON has no MIME field for ordinary downloaded media. Restricting this
/// inference to the explicit image extensions keeps binary video/audio and
/// unknown extensions out of multimodal summary input.
fn tweet_media_image_mime(extension: Option<&str>) -> Option<String> {
match extension?.to_ascii_lowercase().as_str() {
"jpg" | "jpeg" => Some("image/jpeg".to_string()),
"png" => Some("image/png".to_string()),
"webp" => Some("image/webp".to_string()),
"gif" => Some("image/gif".to_string()),
"avif" => Some("image/avif".to_string()),
_ => None,
}
}
fn tweet_raw_artifacts(tweet_json: &str) -> Result<Vec<(String, String)>> { fn tweet_raw_artifacts(tweet_json: &str) -> Result<Vec<(String, String)>> {
let regex = regex::Regex::new(r#""(avatar_local_path|local_path)": "([^"\n]+)""#)?; let regex = regex::Regex::new(r#""(avatar_local_path|local_path)": "([^"\n]+)""#)?;
let mut seen = HashSet::new(); let mut seen = HashSet::new();
@ -2810,7 +2829,11 @@ mod tests {
archive::initialize_store_directories(&store_path).unwrap(); archive::initialize_store_directories(&store_path).unwrap();
fs::create_dir_all(&archive_path).unwrap(); fs::create_dir_all(&archive_path).unwrap();
fs::write(archive_path.join("name"), "test-archive").unwrap(); fs::write(archive_path.join("name"), "test-archive").unwrap();
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); fs::write(
archive_path.join("store_path"),
store_path.to_str().unwrap(),
)
.unwrap();
let archive_paths = ArchivePaths { let archive_paths = ArchivePaths {
archive_path: archive_path.clone(), archive_path: archive_path.clone(),
@ -2832,7 +2855,8 @@ mod tests {
// Verify entry was created // Verify entry was created
let conn = database::open_or_initialize(&archive_path).unwrap(); let conn = database::open_or_initialize(&archive_path).unwrap();
let default_coll_id = database::ensure_default_collection(&conn).unwrap(); let default_coll_id = database::ensure_default_collection(&conn).unwrap();
let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap(); let entries =
archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
assert_eq!(entries.len(), 1); assert_eq!(entries.len(), 1);
let entry = &entries[0]; let entry = &entries[0];
assert_eq!(entry.title, Some(title.to_string())); assert_eq!(entry.title, Some(title.to_string()));
@ -2859,7 +2883,11 @@ mod tests {
archive::initialize_store_directories(&store_path).unwrap(); archive::initialize_store_directories(&store_path).unwrap();
fs::create_dir_all(&archive_path).unwrap(); fs::create_dir_all(&archive_path).unwrap();
fs::write(archive_path.join("name"), "test-archive").unwrap(); fs::write(archive_path.join("name"), "test-archive").unwrap();
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); fs::write(
archive_path.join("store_path"),
store_path.to_str().unwrap(),
)
.unwrap();
let archive_paths = ArchivePaths { let archive_paths = ArchivePaths {
archive_path: archive_path.clone(), archive_path: archive_path.clone(),
@ -2878,7 +2906,8 @@ mod tests {
// Verify entry was created // Verify entry was created
let conn = database::open_or_initialize(&archive_path).unwrap(); let conn = database::open_or_initialize(&archive_path).unwrap();
let default_coll_id = database::ensure_default_collection(&conn).unwrap(); let default_coll_id = database::ensure_default_collection(&conn).unwrap();
let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap(); let entries =
archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
assert_eq!(entries.len(), 1); assert_eq!(entries.len(), 1);
let entry = &entries[0]; let entry = &entries[0];
assert_eq!(entry.title, Some(title.to_string())); assert_eq!(entry.title, Some(title.to_string()));
@ -2901,7 +2930,11 @@ mod tests {
archive::initialize_store_directories(&store_path).unwrap(); archive::initialize_store_directories(&store_path).unwrap();
fs::create_dir_all(&archive_path).unwrap(); fs::create_dir_all(&archive_path).unwrap();
fs::write(archive_path.join("name"), "test-archive").unwrap(); fs::write(archive_path.join("name"), "test-archive").unwrap();
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); fs::write(
archive_path.join("store_path"),
store_path.to_str().unwrap(),
)
.unwrap();
let archive_paths = ArchivePaths { let archive_paths = ArchivePaths {
archive_path, archive_path,
@ -2911,7 +2944,10 @@ mod tests {
let result = perform_text_capture(&archive_paths, "", "Some body", "text/plain", None); let result = perform_text_capture(&archive_paths, "", "Some body", "text/plain", None);
assert!(result.is_err()); assert!(result.is_err());
assert!(result.unwrap_err().to_string().contains("title must not be empty")); assert!(result
.unwrap_err()
.to_string()
.contains("title must not be empty"));
// Clean up // Clean up
let _ = fs::remove_dir_all(&base_path); let _ = fs::remove_dir_all(&base_path);
@ -2931,7 +2967,11 @@ mod tests {
archive::initialize_store_directories(&store_path).unwrap(); archive::initialize_store_directories(&store_path).unwrap();
fs::create_dir_all(&archive_path).unwrap(); fs::create_dir_all(&archive_path).unwrap();
fs::write(archive_path.join("name"), "test-archive").unwrap(); fs::write(archive_path.join("name"), "test-archive").unwrap();
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); fs::write(
archive_path.join("store_path"),
store_path.to_str().unwrap(),
)
.unwrap();
let archive_paths = ArchivePaths { let archive_paths = ArchivePaths {
archive_path, archive_path,
@ -2941,7 +2981,10 @@ mod tests {
let result = perform_text_capture(&archive_paths, "Some Title", "", "text/plain", None); let result = perform_text_capture(&archive_paths, "Some Title", "", "text/plain", None);
assert!(result.is_err()); assert!(result.is_err());
assert!(result.unwrap_err().to_string().contains("body must not be empty")); assert!(result
.unwrap_err()
.to_string()
.contains("body must not be empty"));
// Clean up // Clean up
let _ = fs::remove_dir_all(&base_path); let _ = fs::remove_dir_all(&base_path);
@ -2961,7 +3004,11 @@ mod tests {
archive::initialize_store_directories(&store_path).unwrap(); archive::initialize_store_directories(&store_path).unwrap();
fs::create_dir_all(&archive_path).unwrap(); fs::create_dir_all(&archive_path).unwrap();
fs::write(archive_path.join("name"), "test-archive").unwrap(); fs::write(archive_path.join("name"), "test-archive").unwrap();
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); fs::write(
archive_path.join("store_path"),
store_path.to_str().unwrap(),
)
.unwrap();
let archive_paths = ArchivePaths { let archive_paths = ArchivePaths {
archive_path, archive_path,
@ -2969,15 +3016,12 @@ mod tests {
name: "test-archive".to_string(), name: "test-archive".to_string(),
}; };
let result = perform_text_capture( let result = perform_text_capture(&archive_paths, "Title", "Body", "text/html", None);
&archive_paths,
"Title",
"Body",
"text/html",
None,
);
assert!(result.is_err()); assert!(result.is_err());
assert!(result.unwrap_err().to_string().contains("unsupported MIME type")); assert!(result
.unwrap_err()
.to_string()
.contains("unsupported MIME type"));
// Clean up // Clean up
let _ = fs::remove_dir_all(&base_path); let _ = fs::remove_dir_all(&base_path);
@ -3003,12 +3047,15 @@ mod tests {
#[test] #[test]
fn test_record_tweet_entry_links_json_and_raw_artifacts() { fn test_record_tweet_entry_links_json_and_raw_artifacts() {
let store_path = env::temp_dir().join(format!( let temp = tempfile::tempdir().unwrap();
"archivr-tweet-db-test-{}", let archive_paths = archive::initialize_archive(
Local::now().format("%Y%m%d%H%M%S%3f") temp.path(),
)); &temp.path().join("store"),
let _ = fs::remove_dir_all(&store_path); "Tweet summary test",
archive::initialize_store_directories(&store_path).unwrap(); false,
)
.unwrap();
let store_path = &archive_paths.store_path;
fs::create_dir_all(store_path.join("raw").join("a").join("b")).unwrap(); fs::create_dir_all(store_path.join("raw").join("a").join("b")).unwrap();
fs::create_dir_all(store_path.join("raw").join("c").join("d")).unwrap(); fs::create_dir_all(store_path.join("raw").join("c").join("d")).unwrap();
fs::write( fs::write(
@ -3025,21 +3072,21 @@ mod tests {
.join("raw") .join("raw")
.join("c") .join("c")
.join("d") .join("d")
.join("cdef01.mp4"), .join("cdef01.jpg"),
b"media", b"media",
) )
.unwrap(); .unwrap();
fs::write( fs::write(
store_path.join("raw_tweets").join("tweet-123.json"), store_path.join("raw_tweets").join("tweet-123.json"),
r#"{ r#"{
"full_text": "Tweet body for summary selection.",
"author": { "avatar_local_path": "raw/a/b/abcdef.jpg" }, "author": { "avatar_local_path": "raw/a/b/abcdef.jpg" },
"entities": { "media": [{ "local_path": "raw/c/d/cdef01.mp4" }] } "entities": { "media": [{ "local_path": "raw/c/d/cdef01.jpg" }] }
}"#, }"#,
) )
.unwrap(); .unwrap();
let conn = rusqlite::Connection::open_in_memory().unwrap(); let conn = database::open_or_initialize(&archive_paths.archive_path).unwrap();
database::initialize_schema(&conn).unwrap();
let user_id = database::ensure_default_user(&conn).unwrap(); let user_id = database::ensure_default_user(&conn).unwrap();
let run = database::create_archive_run(&conn, user_id, 1).unwrap(); let run = database::create_archive_run(&conn, user_id, 1).unwrap();
let item = database::create_archive_run_item( let item = database::create_archive_run_item(
@ -3088,10 +3135,26 @@ mod tests {
assert_eq!(artifact_count, 3); assert_eq!(artifact_count, 3);
assert_eq!(blob_count, 2); assert_eq!(blob_count, 2);
let media_mime: Option<String> = conn
.query_row(
"SELECT b.mime_type FROM entry_artifacts ea JOIN blobs b ON b.id = ea.blob_id WHERE ea.entry_id = ?1 AND ea.artifact_role = 'media'",
[entry.id],
|row| row.get(0),
)
.unwrap();
assert_eq!(media_mime.as_deref(), Some("image/jpeg"));
let summary_input = crate::summarizer::build_summary_input(
&archive_paths,
&entry.entry_uid,
crate::summarizer::SummaryBuildOptions {
include_images: true,
},
)
.unwrap();
assert_eq!(summary_input.request.images.len(), 1);
assert_eq!(summary_input.request.images[0].mime_type, "image/jpeg");
assert_eq!(run_status, "completed"); assert_eq!(run_status, "completed");
assert!(store_path.join(&entry.structured_root_relpath).is_dir()); assert!(store_path.join(&entry.structured_root_relpath).is_dir());
let _ = fs::remove_dir_all(store_path);
} }
mod title_tests { mod title_tests {

View file

@ -121,6 +121,8 @@ pub struct EntrySummaryRecord {
pub summary_uid: String, pub summary_uid: String,
pub entry_uid: String, pub entry_uid: String,
pub provider_kind: String, pub provider_kind: String,
/// Model selected by the provider after resolving the requested cache-key alias.
pub resolved_model: Option<String>,
pub provider_model: Option<String>, pub provider_model: Option<String>,
pub prompt_version: String, pub prompt_version: String,
pub input_sha256: String, pub input_sha256: String,
@ -350,6 +352,7 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
entry_id INTEGER NOT NULL REFERENCES archived_entries(id) ON DELETE CASCADE, entry_id INTEGER NOT NULL REFERENCES archived_entries(id) ON DELETE CASCADE,
provider_kind TEXT NOT NULL, provider_kind TEXT NOT NULL,
provider_model TEXT NOT NULL DEFAULT '', provider_model TEXT NOT NULL DEFAULT '',
resolved_model TEXT,
prompt_version TEXT NOT NULL, prompt_version TEXT NOT NULL,
input_sha256 TEXT NOT NULL, input_sha256 TEXT NOT NULL,
status TEXT NOT NULL CHECK(status IN ('pending','running','completed','failed')), status TEXT NOT NULL CHECK(status IN ('pending','running','completed','failed')),
@ -477,6 +480,12 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
// Migration: add notes_json column to existing capture_jobs tables. // Migration: add notes_json column to existing capture_jobs tables.
// Silently ignored when the column already exists (idempotent). // Silently ignored when the column already exists (idempotent).
let _ = conn.execute("ALTER TABLE capture_jobs ADD COLUMN notes_json TEXT", []); let _ = conn.execute("ALTER TABLE capture_jobs ADD COLUMN notes_json TEXT", []);
// Provider responses may resolve a requested alias to a concrete model.
// Keep that display-only value outside the cache key.
let _ = conn.execute(
"ALTER TABLE entry_summaries ADD COLUMN resolved_model TEXT",
[],
);
// Migration: add requires_auth column to existing collections tables. // Migration: add requires_auth column to existing collections tables.
// Silently ignored when the column already exists (idempotent). // Silently ignored when the column already exists (idempotent).
let _ = conn.execute( let _ = conn.execute(
@ -494,10 +503,11 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
|row| row.get(0), |row| row.get(0),
) )
.optional()?; .optional()?;
if summary_table_sql if summary_table_sql.as_deref().is_some_and(|sql| {
.as_deref() sql.contains(
.is_some_and(|sql| sql.contains("UNIQUE(entry_id, provider_kind, provider_model, prompt_version, input_sha256)")) "UNIQUE(entry_id, provider_kind, provider_model, prompt_version, input_sha256)",
{ )
}) {
conn.execute_batch( conn.execute_batch(
"BEGIN; "BEGIN;
CREATE TABLE entry_summaries_rebuilt ( CREATE TABLE entry_summaries_rebuilt (
@ -506,6 +516,7 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
entry_id INTEGER NOT NULL REFERENCES archived_entries(id) ON DELETE CASCADE, entry_id INTEGER NOT NULL REFERENCES archived_entries(id) ON DELETE CASCADE,
provider_kind TEXT NOT NULL, provider_kind TEXT NOT NULL,
provider_model TEXT NOT NULL DEFAULT '', provider_model TEXT NOT NULL DEFAULT '',
resolved_model TEXT,
prompt_version TEXT NOT NULL, prompt_version TEXT NOT NULL,
input_sha256 TEXT NOT NULL, input_sha256 TEXT NOT NULL,
status TEXT NOT NULL CHECK(status IN ('pending','running','completed','failed')), status TEXT NOT NULL CHECK(status IN ('pending','running','completed','failed')),
@ -517,7 +528,7 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
); );
INSERT INTO entry_summaries_rebuilt INSERT INTO entry_summaries_rebuilt
SELECT id, summary_uid, entry_id, provider_kind, COALESCE(provider_model, ''), SELECT id, summary_uid, entry_id, provider_kind, COALESCE(provider_model, ''),
prompt_version, input_sha256, status, summary_text, error_text, resolved_model, prompt_version, input_sha256, status, summary_text, error_text,
created_at, updated_at, completed_at created_at, updated_at, completed_at
FROM entry_summaries; FROM entry_summaries;
DROP TABLE entry_summaries; DROP TABLE entry_summaries;
@ -1465,7 +1476,8 @@ pub fn fail_stalled_entry_summaries(conn: &Connection) -> Result<usize> {
/// `SELECT` list shared by every `entry_summaries` read, so all readers build /// `SELECT` list shared by every `entry_summaries` read, so all readers build
/// an identical `EntrySummaryRecord` from the same column ordering. /// an identical `EntrySummaryRecord` from the same column ordering.
const ENTRY_SUMMARY_COLS: &str = "SELECT s.summary_uid, e.entry_uid, s.provider_kind, s.provider_model, const ENTRY_SUMMARY_COLS: &str =
"SELECT s.summary_uid, e.entry_uid, s.provider_kind, s.provider_model, s.resolved_model,
s.prompt_version, s.input_sha256, s.status, s.summary_text, s.prompt_version, s.input_sha256, s.status, s.summary_text,
s.error_text, s.created_at, s.updated_at, s.completed_at s.error_text, s.created_at, s.updated_at, s.completed_at
FROM entry_summaries s FROM entry_summaries s
@ -1478,17 +1490,16 @@ fn map_entry_summary(row: &rusqlite::Row<'_>) -> rusqlite::Result<EntrySummaryRe
provider_kind: row.get(2)?, provider_kind: row.get(2)?,
// '' is the stored stand-in for "this provider has no model"; see the // '' is the stored stand-in for "this provider has no model"; see the
// doc comment on EntrySummaryRecord for why it is not NULL. // doc comment on EntrySummaryRecord for why it is not NULL.
provider_model: row provider_model: row.get::<_, Option<String>>(3)?.filter(|m| !m.is_empty()),
.get::<_, Option<String>>(3)? resolved_model: row.get(4)?,
.filter(|m| !m.is_empty()), prompt_version: row.get(5)?,
prompt_version: row.get(4)?, input_sha256: row.get(6)?,
input_sha256: row.get(5)?, status: row.get(7)?,
status: row.get(6)?, summary_text: row.get(8)?,
summary_text: row.get(7)?, error_text: row.get(9)?,
error_text: row.get(8)?, created_at: row.get(10)?,
created_at: row.get(9)?, updated_at: row.get(11)?,
updated_at: row.get(10)?, completed_at: row.get(12)?,
completed_at: row.get(11)?,
}) })
} }
@ -1536,6 +1547,25 @@ pub fn upsert_pending_entry_summary(
Ok(summary_uid) Ok(summary_uid)
} }
/// Completes a summary while retaining the provider's concrete response model
/// for display. The requested provider model remains the cache-key identity.
pub fn update_entry_summary_completed(
conn: &Connection,
summary_uid: &str,
summary_text: &str,
resolved_model: Option<&str>,
) -> Result<()> {
let now = now_timestamp();
conn.execute(
"UPDATE entry_summaries
SET status = 'completed', summary_text = ?1, error_text = NULL,
resolved_model = ?2, completed_at = ?3, updated_at = ?3
WHERE summary_uid = ?4",
rusqlite::params![summary_text, resolved_model, now, summary_uid],
)?;
Ok(())
}
/// Moves a summary row through `running` → `completed` / `failed`. /// Moves a summary row through `running` → `completed` / `failed`.
/// `completed_at` is stamped only on the terminal `completed` transition. /// `completed_at` is stamped only on the terminal `completed` transition.
pub fn update_entry_summary_status( pub fn update_entry_summary_status(
@ -1556,7 +1586,14 @@ pub fn update_entry_summary_status(
SET status = ?1, summary_text = ?2, error_text = ?3, SET status = ?1, summary_text = ?2, error_text = ?3,
completed_at = ?4, updated_at = ?5 completed_at = ?4, updated_at = ?5
WHERE summary_uid = ?6", WHERE summary_uid = ?6",
rusqlite::params![status, summary_text, error_text, completed_at, now, summary_uid], rusqlite::params![
status,
summary_text,
error_text,
completed_at,
now,
summary_uid
],
)?; )?;
Ok(()) Ok(())
} }
@ -1733,7 +1770,11 @@ pub fn finish_archive_run(conn: &Connection, run_id: i64) -> Result<()> {
[run_id], [run_id],
|row| row.get(0), |row| row.get(0),
)?; )?;
let status = if failed_count > 0 { "failed" } else { "completed" }; let status = if failed_count > 0 {
"failed"
} else {
"completed"
};
conn.execute( conn.execute(
"UPDATE archive_runs SET status = ?1, finished_at = ?2 WHERE id = ?3", "UPDATE archive_runs SET status = ?1, finished_at = ?2 WHERE id = ?3",
params![status, now_timestamp(), run_id], params![status, now_timestamp(), run_id],
@ -1969,9 +2010,8 @@ pub fn delete_entry(conn: &Connection, entry_uid: &str) -> Result<bool> {
// (no grandchildren), so without `id = ?1` the set would be empty and // (no grandchildren), so without `id = ?1` the set would be empty and
// cascade_cached_bytes_after_subtree_delete would not recalculate shared-blob totals. // cascade_cached_bytes_after_subtree_delete would not recalculate shared-blob totals.
let subtree_ids: Vec<i64> = { let subtree_ids: Vec<i64> = {
let mut stmt = conn.prepare( let mut stmt =
"SELECT id FROM archived_entries WHERE id = ?1 OR root_entry_id = ?1", conn.prepare("SELECT id FROM archived_entries WHERE id = ?1 OR root_entry_id = ?1")?;
)?;
stmt.query_map([entry_id], |row| row.get(0))? stmt.query_map([entry_id], |row| row.get(0))?
.collect::<rusqlite::Result<_>>()? .collect::<rusqlite::Result<_>>()?
}; };
@ -2622,7 +2662,6 @@ pub fn visibility_to_bits(visibility: &str) -> u32 {
} }
} }
/// Returns the id of the '_default_' collection, creating it if absent. /// Returns the id of the '_default_' collection, creating it if absent.
pub fn ensure_default_collection(conn: &Connection) -> Result<i64> { pub fn ensure_default_collection(conn: &Connection) -> Result<i64> {
let now = now_timestamp(); let now = now_timestamp();
@ -2725,7 +2764,8 @@ pub fn get_collection_by_slug(conn: &Connection, slug: &str) -> Result<Option<Co
"SELECT id, collection_uid, name, slug, default_visibility_bits, created_at, requires_auth \ "SELECT id, collection_uid, name, slug, default_visibility_bits, created_at, requires_auth \
FROM collections WHERE slug = ?1", FROM collections WHERE slug = ?1",
[slug], [slug],
|row| Ok(CollectionRecord { |row| {
Ok(CollectionRecord {
id: row.get(0)?, id: row.get(0)?,
collection_uid: row.get(1)?, collection_uid: row.get(1)?,
name: row.get(2)?, name: row.get(2)?,
@ -2733,8 +2773,11 @@ pub fn get_collection_by_slug(conn: &Connection, slug: &str) -> Result<Option<Co
default_visibility_bits: row.get::<_, i64>(4)? as u32, default_visibility_bits: row.get::<_, i64>(4)? as u32,
created_at: row.get(5)?, created_at: row.get(5)?,
requires_auth: row.get::<_, i64>(6)? != 0, requires_auth: row.get::<_, i64>(6)? != 0,
}), })
).optional().map_err(Into::into) },
)
.optional()
.map_err(Into::into)
} }
/// Adds an entry to a collection with given visibility_bits. Idempotent (INSERT OR IGNORE). /// Adds an entry to a collection with given visibility_bits. Idempotent (INSERT OR IGNORE).
@ -2793,7 +2836,12 @@ pub fn get_entry_collection_memberships(
WHERE ce.entry_id = ?1", WHERE ce.entry_id = ?1",
)?; )?;
stmt.query_map([entry_id], |row| { stmt.query_map([entry_id], |row| {
Ok((row.get(0)?, row.get(1)?, row.get(2)?, row.get::<_, i64>(3)? as u32)) Ok((
row.get(0)?,
row.get(1)?,
row.get(2)?,
row.get::<_, i64>(3)? as u32,
))
})? })?
.collect::<Result<_, _>>() .collect::<Result<_, _>>()
.map_err(Into::into) .map_err(Into::into)
@ -3351,7 +3399,10 @@ mod tests {
|row| row.get::<_, i64>(0).map(|v| v as u32), |row| row.get::<_, i64>(0).map(|v| v as u32),
) )
.unwrap(); .unwrap();
assert_eq!(default_bits, 2, "default collection should start USER-visible"); assert_eq!(
default_bits, 2,
"default collection should start USER-visible"
);
// Create an entry with visibility = "private" (what capture always passes). // Create an entry with visibility = "private" (what capture always passes).
let entry = create_entry_fixture(&conn, "private", None, None); let entry = create_entry_fixture(&conn, "private", None, None);
@ -3399,7 +3450,10 @@ mod tests {
|row| row.get::<_, i64>(0).map(|v| v as u32), |row| row.get::<_, i64>(0).map(|v| v as u32),
) )
.unwrap(); .unwrap();
assert_eq!(public_bits, 3, "collection default=public should produce bits=3"); assert_eq!(
public_bits, 3,
"collection default=public should produce bits=3"
);
// Child entries must NOT use the collection default — they keep // Child entries must NOT use the collection default — they keep
// visibility_to_bits(entry.visibility) so parent-child visibility // visibility_to_bits(entry.visibility) so parent-child visibility
@ -3412,7 +3466,10 @@ mod tests {
|row| row.get::<_, i64>(0).map(|v| v as u32), |row| row.get::<_, i64>(0).map(|v| v as u32),
) )
.unwrap(); .unwrap();
assert_eq!(child_bits, 0, "child entries must use visibility_to_bits, not collection default"); assert_eq!(
child_bits, 0,
"child entries must use visibility_to_bits, not collection default"
);
} }
#[test] #[test]
@ -4441,30 +4498,50 @@ mod tests {
// Root item (parent_item_id IS NULL) — mirrors what record_container_entry does. // Root item (parent_item_id IS NULL) — mirrors what record_container_entry does.
let root_item = create_archive_run_item( let root_item = create_archive_run_item(
&c, run.id, None, 0, "https://example.com/pl", None, "youtube", "playlist", &c,
).unwrap(); run.id,
None,
0,
"https://example.com/pl",
None,
"youtube",
"playlist",
)
.unwrap();
// Child item (parent_item_id IS NOT NULL). // Child item (parent_item_id IS NOT NULL).
let child_item = create_archive_run_item( let child_item = create_archive_run_item(
&c, run.id, Some(root_item.id), 1, "https://example.com/v1", None, "youtube", "video", &c,
).unwrap(); run.id,
Some(root_item.id),
1,
"https://example.com/v1",
None,
"youtube",
"video",
)
.unwrap();
// Complete both — marks archive_runs.completed_count = 2. // Complete both — marks archive_runs.completed_count = 2.
c.execute( c.execute(
"UPDATE archive_run_items SET status = 'completed' WHERE id IN (?1, ?2)", "UPDATE archive_run_items SET status = 'completed' WHERE id IN (?1, ?2)",
rusqlite::params![root_item.id, child_item.id], rusqlite::params![root_item.id, child_item.id],
).unwrap(); )
.unwrap();
refresh_run_counters(&c, run.id).unwrap(); refresh_run_counters(&c, run.id).unwrap();
let total: i64 = c.query_row( let total: i64 = c
"SELECT completed_count FROM archive_runs WHERE id = ?1", [run.id], |r| r.get(0), .query_row(
).unwrap(); "SELECT completed_count FROM archive_runs WHERE id = ?1",
[run.id],
|r| r.get(0),
)
.unwrap();
assert_eq!(total, 2, "both items completed: DB counter must be 2"); assert_eq!(total, 2, "both items completed: DB counter must be 2");
let child_count = get_run_completed_child_count(&c, run.id).unwrap(); let child_count = get_run_completed_child_count(&c, run.id).unwrap();
assert_eq!(child_count, 1, "only the child item must be counted"); assert_eq!(child_count, 1, "only the child item must be counted");
} }
// ── entry_summaries ──────────────────────────────────────────────────── // ── entry_summaries ────────────────────────────────────────────────────
#[test] #[test]
@ -4489,6 +4566,41 @@ mod tests {
assert!(latest_entry_summary(&c, entry.id).unwrap().is_some()); assert!(latest_entry_summary(&c, entry.id).unwrap().is_some());
} }
#[test]
fn initialize_schema_migrates_resolved_model_for_existing_summary_table() {
let c = Connection::open_in_memory().unwrap();
c.execute_batch(
"CREATE TABLE entry_summaries (
id INTEGER PRIMARY KEY,
summary_uid TEXT NOT NULL UNIQUE,
entry_id INTEGER NOT NULL,
provider_kind TEXT NOT NULL,
provider_model TEXT,
prompt_version TEXT NOT NULL,
input_sha256 TEXT NOT NULL,
status TEXT NOT NULL,
summary_text TEXT,
error_text TEXT,
created_at TEXT NOT NULL,
updated_at TEXT NOT NULL,
completed_at TEXT,
UNIQUE(entry_id, provider_kind, provider_model, prompt_version, input_sha256)
);",
)
.unwrap();
initialize_schema(&c).unwrap();
let has_resolved_model: i64 = c
.query_row(
"SELECT COUNT(*) FROM pragma_table_info('entry_summaries') WHERE name = 'resolved_model'",
[],
|r| r.get(0),
)
.unwrap();
assert_eq!(has_resolved_model, 1);
}
#[test] #[test]
fn entry_summary_lifecycle_pending_running_completed() { fn entry_summary_lifecycle_pending_running_completed() {
let c = conn(); let c = conn();
@ -4519,6 +4631,48 @@ mod tests {
assert!(rec.completed_at.is_some(), "completed rows must be stamped"); assert!(rec.completed_at.is_some(), "completed rows must be stamped");
} }
#[test]
fn entry_summary_keeps_requested_model_for_cache_and_resolved_model_for_display() {
let c = conn();
let entry = create_entry_fixture(&c, "private", None, None);
let uid = upsert_pending_entry_summary(
&c,
entry.id,
"anthropic_http",
Some("claude-3-5-sonnet-latest"),
"v1",
"sha1",
)
.unwrap();
update_entry_summary_completed(&c, &uid, "summary", Some("claude-3-5-sonnet-20241022"))
.unwrap();
let rec = get_entry_summary_by_uid(&c, &uid).unwrap().unwrap();
assert_eq!(
rec.provider_model.as_deref(),
Some("claude-3-5-sonnet-latest")
);
assert_eq!(
rec.resolved_model.as_deref(),
Some("claude-3-5-sonnet-20241022")
);
assert_eq!(
find_entry_summary(
&c,
entry.id,
"anthropic_http",
Some("claude-3-5-sonnet-latest"),
"v1",
"sha1",
)
.unwrap()
.unwrap()
.summary_uid,
uid,
);
}
#[test] #[test]
fn entry_summary_failure_records_error_and_no_completed_at() { fn entry_summary_failure_records_error_and_no_completed_at() {
let c = conn(); let c = conn();
@ -4569,12 +4723,12 @@ mod tests {
fn fail_stalled_entry_summaries_marks_pending_and_running_with_restart_message() { fn fail_stalled_entry_summaries_marks_pending_and_running_with_restart_message() {
let c = conn(); let c = conn();
let entry = create_entry_fixture(&c, "private", None, None); let entry = create_entry_fixture(&c, "private", None, None);
let pending = upsert_pending_entry_summary(&c, entry.id, "claude_cli", None, "v1", "a") let pending =
.unwrap(); upsert_pending_entry_summary(&c, entry.id, "claude_cli", None, "v1", "a").unwrap();
let running = upsert_pending_entry_summary(&c, entry.id, "codex_cli", None, "v1", "b") let running =
.unwrap(); upsert_pending_entry_summary(&c, entry.id, "codex_cli", None, "v1", "b").unwrap();
let completed = upsert_pending_entry_summary(&c, entry.id, "anthropic_http", None, "v1", "c") let completed =
.unwrap(); upsert_pending_entry_summary(&c, entry.id, "anthropic_http", None, "v1", "c").unwrap();
update_entry_summary_status(&c, &running, "running", None, None).unwrap(); update_entry_summary_status(&c, &running, "running", None, None).unwrap();
update_entry_summary_status(&c, &completed, "completed", Some("done"), None).unwrap(); update_entry_summary_status(&c, &completed, "completed", Some("done"), None).unwrap();
@ -4589,7 +4743,10 @@ mod tests {
assert!(!row.updated_at.is_empty()); assert!(!row.updated_at.is_empty());
} }
assert_eq!( assert_eq!(
get_entry_summary_by_uid(&c, &completed).unwrap().unwrap().status, get_entry_summary_by_uid(&c, &completed)
.unwrap()
.unwrap()
.status,
"completed" "completed"
); );
} }
@ -4637,7 +4794,14 @@ mod tests {
.unwrap(); .unwrap();
assert_eq!(hit.summary_uid, uid); assert_eq!(hit.summary_uid, uid);
assert!( assert!(
find_entry_summary(&c, entry.id, "openai_compatible", Some("gpt"), "v1", "changed") find_entry_summary(
&c,
entry.id,
"openai_compatible",
Some("gpt"),
"v1",
"changed"
)
.unwrap() .unwrap()
.is_none(), .is_none(),
"a changed input digest must miss the cache" "a changed input digest must miss the cache"
@ -4653,7 +4817,13 @@ mod tests {
update_entry_summary_status(&c, &a, "completed", Some("first"), None).unwrap(); update_entry_summary_status(&c, &a, "completed", Some("first"), None).unwrap();
update_entry_summary_status(&c, &b, "completed", Some("second"), None).unwrap(); update_entry_summary_status(&c, &b, "completed", Some("second"), None).unwrap();
// Same-timestamp ties break on id DESC, so the later insert wins either way. // Same-timestamp ties break on id DESC, so the later insert wins either way.
assert_eq!(latest_entry_summary(&c, entry.id).unwrap().unwrap().summary_uid, b); assert_eq!(
latest_entry_summary(&c, entry.id)
.unwrap()
.unwrap()
.summary_uid,
b
);
} }
#[test] #[test]
@ -4680,7 +4850,10 @@ mod tests {
fn entry_id_for_uid_resolves_and_misses() { fn entry_id_for_uid_resolves_and_misses() {
let c = conn(); let c = conn();
let entry = create_entry_fixture(&c, "private", None, None); let entry = create_entry_fixture(&c, "private", None, None);
assert_eq!(entry_id_for_uid(&c, &entry.entry_uid).unwrap(), Some(entry.id)); assert_eq!(
entry_id_for_uid(&c, &entry.entry_uid).unwrap(),
Some(entry.id)
);
assert_eq!(entry_id_for_uid(&c, "ent_nope").unwrap(), None); assert_eq!(entry_id_for_uid(&c, "ent_nope").unwrap(), None);
} }
} }

View file

@ -253,13 +253,19 @@ fn resolve_cli(env_name: &str, well_known_absolute: &[&str], bare: &str) -> Path
pub fn provider_from_env(kind: &str) -> Result<ProviderConfig> { pub fn provider_from_env(kind: &str) -> Result<ProviderConfig> {
match kind { match kind {
"anthropic_http" => Ok(ProviderConfig::AnthropicHttp(HttpProviderConfig { "anthropic_http" => Ok(ProviderConfig::AnthropicHttp(HttpProviderConfig {
endpoint: env_or("ARCHIVR_ANTHROPIC_URL", "https://api.anthropic.com/v1/messages"), endpoint: env_or(
"ARCHIVR_ANTHROPIC_URL",
"https://api.anthropic.com/v1/messages",
),
api_key: required_env("ARCHIVR_ANTHROPIC_API_KEY")?, api_key: required_env("ARCHIVR_ANTHROPIC_API_KEY")?,
model: env_or("ARCHIVR_ANTHROPIC_MODEL", "claude-3-5-sonnet-latest"), model: env_or("ARCHIVR_ANTHROPIC_MODEL", "claude-3-5-sonnet-latest"),
timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS), timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS),
})), })),
"openai_compatible" => Ok(ProviderConfig::OpenAiCompatible(HttpProviderConfig { "openai_compatible" => Ok(ProviderConfig::OpenAiCompatible(HttpProviderConfig {
endpoint: env_or("ARCHIVR_OPENAI_URL", "https://api.openai.com/v1/chat/completions"), endpoint: env_or(
"ARCHIVR_OPENAI_URL",
"https://api.openai.com/v1/chat/completions",
),
api_key: required_env("ARCHIVR_OPENAI_API_KEY")?, api_key: required_env("ARCHIVR_OPENAI_API_KEY")?,
model: env_or("ARCHIVR_OPENAI_MODEL", "gpt-4o-mini"), model: env_or("ARCHIVR_OPENAI_MODEL", "gpt-4o-mini"),
timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS), timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS),
@ -328,8 +334,12 @@ fn http_client(timeout_secs: u64) -> Result<reqwest::blocking::Client> {
/// Body builder kept separate from the transport so it can be unit-tested /// Body builder kept separate from the transport so it can be unit-tested
/// without a network round-trip. /// without a network round-trip.
fn read_image_base64(image: &SummaryImage) -> Result<String> { fn read_image_base64(image: &SummaryImage) -> Result<String> {
let bytes = std::fs::read(&image.archive_file) let bytes = std::fs::read(&image.archive_file).with_context(|| {
.with_context(|| format!("failed to read summary image {}", image.archive_file.display()))?; format!(
"failed to read summary image {}",
image.archive_file.display()
)
})?;
Ok(base64::engine::general_purpose::STANDARD.encode(bytes)) Ok(base64::engine::general_purpose::STANDARD.encode(bytes))
} }
@ -420,7 +430,10 @@ impl SummaryProvider for AnthropicHttpProvider {
let status = resp.status(); let status = resp.status();
let text = resp.text().unwrap_or_default(); let text = resp.text().unwrap_or_default();
if !status.is_success() { if !status.is_success() {
bail!("anthropic API returned {status}: {}", truncate_for_error(&text)); bail!(
"anthropic API returned {status}: {}",
truncate_for_error(&text)
);
} }
parse_anthropic_response(&text) parse_anthropic_response(&text)
} }
@ -499,12 +512,7 @@ fn truncate_for_error(s: &str) -> String {
/// the child if it overruns. stdin is written on a third thread because a /// the child if it overruns. stdin is written on a third thread because a
/// 48 KB prompt can exceed the pipe buffer, and writing it inline would /// 48 KB prompt can exceed the pipe buffer, and writing it inline would
/// deadlock against a child that is waiting for us to read its output. /// deadlock against a child that is waiting for us to read its output.
fn run_cli( fn run_cli(executable: &Path, args: &[&str], prompt: &str, timeout_secs: u64) -> Result<String> {
executable: &Path,
args: &[&str],
prompt: &str,
timeout_secs: u64,
) -> Result<String> {
let mut child = Command::new(executable) let mut child = Command::new(executable)
.args(args) .args(args)
.stdin(Stdio::piped()) .stdin(Stdio::piped())
@ -538,14 +546,13 @@ fn run_cli(
}); });
let collected = match rx.recv_timeout(Duration::from_secs(timeout_secs)) { let collected = match rx.recv_timeout(Duration::from_secs(timeout_secs)) {
Ok(res) => res.with_context(|| format!("failed to read stdout of {}", executable.display()))?, Ok(res) => {
res.with_context(|| format!("failed to read stdout of {}", executable.display()))?
}
Err(_) => { Err(_) => {
let _ = child.kill(); let _ = child.kill();
let _ = child.wait(); let _ = child.wait();
bail!( bail!("{} timed out after {timeout_secs}s", executable.display());
"{} timed out after {timeout_secs}s",
executable.display()
);
} }
}; };
@ -588,7 +595,12 @@ impl SummaryProvider for ClaudeCliProvider {
args.push("--model"); args.push("--model");
args.push(model); args.push(model);
} }
let out = run_cli(&self.0.executable, &args, &build_combined_prompt(request), self.0.timeout_secs)?; let out = run_cli(
&self.0.executable,
&args,
&build_combined_prompt(request),
self.0.timeout_secs,
)?;
Ok(SummaryOutput { Ok(SummaryOutput {
text: out, text: out,
model: self.0.model.clone(), model: self.0.model.clone(),
@ -638,7 +650,11 @@ mod codex {
let out = std::fs::read_to_string(path).ok()?; let out = std::fs::read_to_string(path).ok()?;
let _ = std::fs::remove_file(path); let _ = std::fs::remove_file(path);
let trimmed = out.trim(); let trimmed = out.trim();
if trimmed.is_empty() { None } else { Some(trimmed.to_string()) } if trimmed.is_empty() {
None
} else {
Some(trimmed.to_string())
}
} }
pub fn primary_args( pub fn primary_args(
@ -698,30 +714,21 @@ mod codex {
// Fallback: positional prompt, no stdin, same --output-last-message. // Fallback: positional prompt, no stdin, same --output-last-message.
let fb = positional_args(images, &out_path, cfg.model.as_deref(), prompt); let fb = positional_args(images, &out_path, cfg.model.as_deref(), prompt);
let out = Command::new(&cfg.executable) let fb_refs: Vec<&str> = fb.iter().map(String::as_str).collect();
.args(&fb) let out = run_cli(&cfg.executable, &fb_refs, "", cfg.timeout_secs).with_context(|| {
.output() let _ = std::fs::remove_file(&out_path);
.with_context(|| {
format!( format!(
"codex `exec -` failed ({primary_err:#}); positional fallback also failed to spawn{}", "codex `exec -` failed ({primary_err:#}); positional fallback failed{}",
missing_binary_hint(cfg) missing_binary_hint(cfg)
) )
})?; })?;
if !out.status.success() {
let _ = std::fs::remove_file(&out_path);
bail!(
"codex `exec -` failed ({primary_err:#}); positional fallback exited with {}: {}",
out.status,
truncate_for_error(&String::from_utf8_lossy(&out.stderr))
);
}
if let Some(text) = read_and_cleanup(&out_path) { if let Some(text) = read_and_cleanup(&out_path) {
return Ok(text); return Ok(text);
} }
// Last resort — the child succeeded but wrote nothing to the file. Fall // Last resort — the child succeeded but wrote nothing to the file. Fall
// back to raw stdout so the caller has *something* to normalize. // back to raw stdout so the caller has *something* to normalize.
let _ = std::fs::remove_file(&out_path); let _ = std::fs::remove_file(&out_path);
Ok(String::from_utf8_lossy(&out.stdout).to_string()) Ok(out)
} }
} }
@ -771,7 +778,8 @@ pub fn strip_html(html: &str) -> String {
.unwrap(); .unwrap();
let comments = regex::Regex::new(r"(?s)<!--.*?-->").unwrap(); let comments = regex::Regex::new(r"(?s)<!--.*?-->").unwrap();
// Treat block-level closers as line breaks so paragraphs do not run together. // Treat block-level closers as line breaks so paragraphs do not run together.
let breaks = regex::Regex::new(r"(?i)</\s*(p|div|br|li|h[1-6]|tr|section|article)\s*>").unwrap(); let breaks =
regex::Regex::new(r"(?i)</\s*(p|div|br|li|h[1-6]|tr|section|article)\s*>").unwrap();
let tags = regex::Regex::new(r"(?s)<[^>]*>").unwrap(); let tags = regex::Regex::new(r"(?s)<[^>]*>").unwrap();
let spaces = regex::Regex::new(r"[ \t\r\f\v]+").unwrap(); let spaces = regex::Regex::new(r"[ \t\r\f\v]+").unwrap();
let blank_lines = regex::Regex::new(r"\n{3,}").unwrap(); let blank_lines = regex::Regex::new(r"\n{3,}").unwrap();
@ -782,11 +790,7 @@ pub fn strip_html(html: &str) -> String {
let s = tags.replace_all(&s, " "); let s = tags.replace_all(&s, " ");
let s = decode_entities(&s); let s = decode_entities(&s);
let s = spaces.replace_all(&s, " "); let s = spaces.replace_all(&s, " ");
let s = s let s = s.lines().map(str::trim).collect::<Vec<_>>().join("\n");
.lines()
.map(str::trim)
.collect::<Vec<_>>()
.join("\n");
blank_lines.replace_all(&s, "\n\n").trim().to_string() blank_lines.replace_all(&s, "\n\n").trim().to_string()
} }
@ -948,9 +952,12 @@ fn load_summary_image_candidates(
store_path: &Path, store_path: &Path,
entry_id: i64, entry_id: i64,
) -> Result<Vec<SummaryImage>> { ) -> Result<Vec<SummaryImage>> {
let canonical_store = store_path let canonical_store = store_path.canonicalize().with_context(|| {
.canonicalize() format!(
.with_context(|| format!("failed to canonicalize store path: {}", store_path.display()))?; "failed to canonicalize store path: {}",
store_path.display()
)
})?;
let mut stmt = conn.prepare( let mut stmt = conn.prepare(
"SELECT b.sha256, b.mime_type, b.extension, b.byte_size, ea.relpath "SELECT b.sha256, b.mime_type, b.extension, b.byte_size, ea.relpath
FROM entry_artifacts ea FROM entry_artifacts ea
@ -1050,7 +1057,11 @@ pub fn build_summary_input(
// Load every matching artifact in insertion order so a thread summarizes // Load every matching artifact in insertion order so a thread summarizes
// as the whole conversation, not just its first status. // as the whole conversation, not just its first status.
let is_tweetish = matches!(entity_kind.as_str(), "tweet" | "tweet_thread"); let is_tweetish = matches!(entity_kind.as_str(), "tweet" | "tweet_thread");
let primary_role = if is_tweetish { "raw_tweet_json" } else { "primary_media" }; let primary_role = if is_tweetish {
"raw_tweet_json"
} else {
"primary_media"
};
let mut artifacts = load_summary_artifacts(&conn, entry_id, primary_role)?; let mut artifacts = load_summary_artifacts(&conn, entry_id, primary_role)?;
if artifacts.is_empty() && is_tweetish { if artifacts.is_empty() && is_tweetish {
// Older archives may have stored tweet payloads under `primary_media`. // Older archives may have stored tweet payloads under `primary_media`.
@ -1066,8 +1077,11 @@ pub fn build_summary_input(
let ext = extension_of(relpath); let ext = extension_of(relpath);
let mime = mime_opt.clone().unwrap_or_default(); let mime = mime_opt.clone().unwrap_or_default();
let piece = if ext == "md" || ext == "markdown" || ext == "txt" let piece = if ext == "md"
|| mime.starts_with("text/markdown") || mime == "text/plain" || ext == "markdown"
|| ext == "txt"
|| mime.starts_with("text/markdown")
|| mime == "text/plain"
{ {
std::fs::read_to_string(&abs) std::fs::read_to_string(&abs)
.with_context(|| format!("failed to read {}", abs.display()))? .with_context(|| format!("failed to read {}", abs.display()))?
@ -1162,8 +1176,9 @@ pub fn normalize_summary_json(raw: &str) -> String {
/// Owns the whole row lifecycle (`pending` → `running` → `completed`/`failed`) /// Owns the whole row lifecycle (`pending` → `running` → `completed`/`failed`)
/// so a caller running it on a background thread only has to handle the /// so a caller running it on a background thread only has to handle the
/// `Err` case. The `(entry_id, provider_kind, provider_model, prompt_version, /// `Err` case. The `(entry_id, provider_kind, provider_model, prompt_version,
/// input_sha256)` UNIQUE key means re-running against unchanged input reuses the /// input_sha256)` cache key identifies equivalent requests. Each generation is
/// same row rather than accumulating duplicates. /// nevertheless recorded as a distinct attempt, so a forced regeneration cannot
/// hide a prior completed result while the new attempt is pending or running.
pub fn summarize_entry( pub fn summarize_entry(
archive_paths: &ArchivePaths, archive_paths: &ArchivePaths,
entry_uid: &str, entry_uid: &str,
@ -1184,12 +1199,7 @@ pub fn summarize_entry(
prompt_version, prompt_version,
&input.input_sha256, &input.input_sha256,
)?; )?;
summarize_prebuilt_entry( summarize_prebuilt_entry(archive_paths, input, &summary_uid, provider)
archive_paths,
input,
&summary_uid,
provider,
)
} }
/// Runs a previously validated and claimed summary attempt. /// Runs a previously validated and claimed summary attempt.
@ -1209,23 +1219,16 @@ pub fn summarize_prebuilt_entry(
match provider.summarize(&input.request) { match provider.summarize(&input.request) {
Ok(output) => { Ok(output) => {
let text = normalize_summary_json(&output.text); let text = normalize_summary_json(&output.text);
database::update_entry_summary_status( database::update_entry_summary_completed(
&conn, &conn,
summary_uid, summary_uid,
"completed", &text,
Some(&text), output.model.as_deref(),
None,
)?; )?;
} }
Err(e) => { Err(e) => {
let msg = format!("{e:#}"); let msg = format!("{e:#}");
database::update_entry_summary_status( database::update_entry_summary_status(&conn, summary_uid, "failed", None, Some(&msg))?;
&conn,
summary_uid,
"failed",
None,
Some(&msg),
)?;
return Err(e); return Err(e);
} }
} }
@ -1317,7 +1320,9 @@ mod tests {
fn provider_from_env_openai_missing_key_names_the_variable() { fn provider_from_env_openai_missing_key_names_the_variable() {
let _g = ENV_LOCK.lock().unwrap(); let _g = ENV_LOCK.lock().unwrap();
clear_provider_env(); clear_provider_env();
let err = provider_from_env("openai_compatible").unwrap_err().to_string(); let err = provider_from_env("openai_compatible")
.unwrap_err()
.to_string();
assert!(err.contains("ARCHIVR_OPENAI_API_KEY"), "got: {err}"); assert!(err.contains("ARCHIVR_OPENAI_API_KEY"), "got: {err}");
} }
@ -1327,7 +1332,10 @@ mod tests {
clear_provider_env(); clear_provider_env();
unsafe { unsafe {
env::set_var("ARCHIVR_OPENAI_API_KEY", "k"); env::set_var("ARCHIVR_OPENAI_API_KEY", "k");
env::set_var("ARCHIVR_OPENAI_URL", "http://localhost:1234/v1/chat/completions"); env::set_var(
"ARCHIVR_OPENAI_URL",
"http://localhost:1234/v1/chat/completions",
);
env::set_var("ARCHIVR_OPENAI_MODEL", "local-model"); env::set_var("ARCHIVR_OPENAI_MODEL", "local-model");
env::set_var("ARCHIVR_SUMMARY_HTTP_TIMEOUT", "7"); env::set_var("ARCHIVR_SUMMARY_HTTP_TIMEOUT", "7");
} }
@ -1353,7 +1361,10 @@ mod tests {
// file-name-`claude` path. // file-name-`claude` path.
assert!( assert!(
c.executable == PathBuf::from("claude") c.executable == PathBuf::from("claude")
|| c.executable.file_name().map(|f| f == "claude").unwrap_or(false), || c.executable
.file_name()
.map(|f| f == "claude")
.unwrap_or(false),
"unexpected claude executable: {}", "unexpected claude executable: {}",
c.executable.display() c.executable.display()
); );
@ -1368,7 +1379,10 @@ mod tests {
// deliberately opportunistic. // deliberately opportunistic.
assert!( assert!(
c.executable == PathBuf::from("codex") c.executable == PathBuf::from("codex")
|| c.executable.file_name().map(|f| f == "codex").unwrap_or(false), || c.executable
.file_name()
.map(|f| f == "codex")
.unwrap_or(false),
"unexpected codex executable: {}", "unexpected codex executable: {}",
c.executable.display() c.executable.display()
); );
@ -1479,29 +1493,57 @@ mod tests {
.position(|arg| arg == "--output-last-message") .position(|arg| arg == "--output-last-message")
.unwrap(); .unwrap();
assert!(image_at < output_at); assert!(image_at < output_at);
assert_eq!(args[image_at + 1], request.images[0].archive_file.to_string_lossy()); assert_eq!(
args[image_at + 1],
request.images[0].archive_file.to_string_lossy()
);
assert_eq!(args.last().unwrap(), "-"); assert_eq!(args.last().unwrap(), "-");
} }
#[test] #[test]
fn codex_positional_arguments_put_images_before_output_path() { fn codex_positional_arguments_put_images_before_output_path() {
let (_temp, request) = image_request(); let (_temp, request) = image_request();
let args = codex::positional_args( let args =
&request.images, codex::positional_args(&request.images, Path::new("/tmp/output"), None, "prompt");
Path::new("/tmp/output"),
None,
"prompt",
);
let image_at = args.iter().position(|arg| arg == "--image").unwrap(); let image_at = args.iter().position(|arg| arg == "--image").unwrap();
let output_at = args let output_at = args
.iter() .iter()
.position(|arg| arg == "--output-last-message") .position(|arg| arg == "--output-last-message")
.unwrap(); .unwrap();
assert!(image_at < output_at); assert!(image_at < output_at);
assert_eq!(args[image_at + 1], request.images[0].archive_file.to_string_lossy()); assert_eq!(
args[image_at + 1],
request.images[0].archive_file.to_string_lossy()
);
assert_eq!(args.last().unwrap(), "prompt"); assert_eq!(args.last().unwrap(), "prompt");
} }
#[cfg(unix)]
#[test]
fn codex_positional_fallback_honors_cli_timeout() {
use std::os::unix::fs::PermissionsExt;
let temp = tempfile::tempdir().unwrap();
let executable = temp.path().join("codex-fixture.sh");
std::fs::write(
&executable,
"#!/bin/sh\nfor arg in \"$@\"; do [ \"$arg\" = \"-\" ] && exit 1; done\nsleep 30\n",
)
.unwrap();
std::fs::set_permissions(&executable, std::fs::Permissions::from_mode(0o755)).unwrap();
let cfg = CliProviderConfig {
executable,
model: None,
timeout_secs: 1,
};
let started = std::time::Instant::now();
let err = format!("{:#}", codex::run(&cfg, "prompt", &[]).unwrap_err());
assert!(started.elapsed() < Duration::from_secs(5));
assert!(err.contains("timed out after 1s"), "got: {err}");
}
#[test] #[test]
fn claude_rejects_images_before_spawning() { fn claude_rejects_images_before_spawning() {
let (_temp, request) = image_request(); let (_temp, request) = image_request();
@ -1511,12 +1553,16 @@ mod tests {
timeout_secs: 1, timeout_secs: 1,
}); });
let err = provider.summarize(&request).unwrap_err().to_string(); let err = provider.summarize(&request).unwrap_err().to_string();
assert!(err.contains("Claude CLI cannot attach local images"), "got: {err}"); assert!(
err.contains("Claude CLI cannot attach local images"),
"got: {err}"
);
} }
#[test] #[test]
fn parse_anthropic_response_extracts_text_and_model() { fn parse_anthropic_response_extracts_text_and_model() {
let body = r#"{"model":"claude-3-5-sonnet-20241022","content":[{"type":"text","text":"hi"}]}"#; let body =
r#"{"model":"claude-3-5-sonnet-20241022","content":[{"type":"text","text":"hi"}]}"#;
let out = parse_anthropic_response(body).unwrap(); let out = parse_anthropic_response(body).unwrap();
assert_eq!(out.text, "hi"); assert_eq!(out.text, "hi");
assert_eq!(out.model.as_deref(), Some("claude-3-5-sonnet-20241022")); assert_eq!(out.model.as_deref(), Some("claude-3-5-sonnet-20241022"));
@ -1559,7 +1605,10 @@ mod tests {
#[test] #[test]
fn strip_html_decodes_common_entities() { fn strip_html_decodes_common_entities() {
assert_eq!(strip_html("<p>a &amp; b &nbsp;c</p>").replace('\u{a0}', " "), "a & b c"); assert_eq!(
strip_html("<p>a &amp; b &nbsp;c</p>").replace('\u{a0}', " "),
"a & b c"
);
} }
#[test] #[test]
@ -1727,8 +1776,14 @@ mod tests {
.unwrap(); .unwrap();
} }
let input = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()).unwrap(); let input =
assert!(input.request.content.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body.")); build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()).unwrap();
assert!(
input
.request
.content
.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body.")
);
} }
fn summary_image_fixture() -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) { fn summary_image_fixture() -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
@ -1800,19 +1855,12 @@ mod tests {
) )
.unwrap(); .unwrap();
let no_artifact = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()) let no_artifact =
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err(); .unwrap_err();
assert!(is_unsupported_summary_content_error(&no_artifact)); assert!(is_unsupported_summary_content_error(&no_artifact));
add_summary_image_artifact( add_summary_image_artifact(&paths, entry.id, 99, "primary_media", "mp4", "video/mp4", 1);
&paths,
entry.id,
99,
"primary_media",
"mp4",
"video/mp4",
1,
);
let video = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()) let video = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err(); .unwrap_err();
assert!(is_unsupported_summary_content_error(&video)); assert!(is_unsupported_summary_content_error(&video));
@ -1839,18 +1887,28 @@ mod tests {
) )
.unwrap(); .unwrap();
drop(conn); drop(conn);
let empty_text = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()) let empty_text =
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err(); .unwrap_err();
assert!(is_unsupported_summary_content_error(&empty_text)); assert!(is_unsupported_summary_content_error(&empty_text));
std::fs::remove_file(paths.store_path.join(empty_relpath)).unwrap(); std::fs::remove_file(paths.store_path.join(empty_relpath)).unwrap();
let read_error = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()) let read_error =
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
.unwrap_err(); .unwrap_err();
assert!(!is_unsupported_summary_content_error(&read_error)); assert!(!is_unsupported_summary_content_error(&read_error));
assert_eq!(UNSUPPORTED_SUMMARY_CONTENT_MESSAGE, "This entry can’t be summarized yet.\n\nIt doesn’t contain archived text that a summary provider can read. Summaries currently support text notes, web pages, X posts and threads, and X Articles. Video, audio, and image-only entries need a transcript or text source."); assert_eq!(
UNSUPPORTED_SUMMARY_CONTENT_MESSAGE,
"This entry can’t be summarized yet.\n\nIt doesn’t contain archived text that a summary provider can read. Summaries currently support text notes, web pages, X posts and threads, and X Articles. Video, audio, and image-only entries need a transcript or text source."
);
assert!(!is_unsupported_summary_content_error(&anyhow!("provider timeout"))); assert!(!is_unsupported_summary_content_error(&anyhow!(
assert!(!is_unsupported_summary_content_error(&anyhow!("entry not found: {}", entry.entry_uid))); "provider timeout"
)));
assert!(!is_unsupported_summary_content_error(&anyhow!(
"entry not found: {}",
entry.entry_uid
)));
} }
fn add_summary_image_artifact( fn add_summary_image_artifact(
@ -1917,13 +1975,17 @@ mod tests {
let text_only = build_summary_input( let text_only = build_summary_input(
&paths, &paths,
&entry.entry_uid, &entry.entry_uid,
SummaryBuildOptions { include_images: false }, SummaryBuildOptions {
include_images: false,
},
) )
.unwrap(); .unwrap();
let visual = build_summary_input( let visual = build_summary_input(
&paths, &paths,
&entry.entry_uid, &entry.entry_uid,
SummaryBuildOptions { include_images: true }, SummaryBuildOptions {
include_images: true,
},
) )
.unwrap(); .unwrap();
@ -1944,38 +2006,112 @@ mod tests {
"summary-image-09", "summary-image-09",
] ]
); );
assert!(visual assert!(
visual
.request .request
.images .images
.iter() .iter()
.all(|image| image.byte_size <= MAX_SUMMARY_IMAGE_BYTES)); .all(|image| image.byte_size <= MAX_SUMMARY_IMAGE_BYTES)
);
} }
#[test] #[test]
fn summary_image_selection_enforces_aggregate_limit_and_keeps_scanning() { fn summary_image_selection_enforces_aggregate_limit_and_keeps_scanning() {
let (_temp, paths, entry) = summary_image_fixture(); let (_temp, paths, entry) = summary_image_fixture();
add_summary_image_artifact(&paths, entry.id, 0, "media", "jpg", "image/jpeg", 4 * 1024 * 1024); add_summary_image_artifact(
add_summary_image_artifact(&paths, entry.id, 1, "media", "png", "image/png", 4 * 1024 * 1024); &paths,
add_summary_image_artifact(&paths, entry.id, 2, "media", "webp", "image/webp", 4 * 1024 * 1024); entry.id,
0,
"media",
"jpg",
"image/jpeg",
4 * 1024 * 1024,
);
add_summary_image_artifact(
&paths,
entry.id,
1,
"media",
"png",
"image/png",
4 * 1024 * 1024,
);
add_summary_image_artifact(
&paths,
entry.id,
2,
"media",
"webp",
"image/webp",
4 * 1024 * 1024,
);
add_summary_image_artifact(&paths, entry.id, 3, "media", "gif", "image/gif", 1); add_summary_image_artifact(&paths, entry.id, 3, "media", "gif", "image/gif", 1);
let visual = build_summary_input( let visual = build_summary_input(
&paths, &paths,
&entry.entry_uid, &entry.entry_uid,
SummaryBuildOptions { include_images: true }, SummaryBuildOptions {
include_images: true,
},
) )
.unwrap(); .unwrap();
assert_eq!(visual.request.images.len(), 3); assert_eq!(visual.request.images.len(), 3);
assert_eq!( assert_eq!(
visual.request.images.iter().map(|image| image.byte_size).sum::<u64>(), visual
MAX_SUMMARY_IMAGE_TOTAL_BYTES
);
assert!(visual
.request .request
.images .images
.iter() .iter()
.all(|image| image.sha256 != "summary-image-03")); .map(|image| image.byte_size)
.sum::<u64>(),
MAX_SUMMARY_IMAGE_TOTAL_BYTES
);
assert!(
visual
.request
.images
.iter()
.all(|image| image.sha256 != "summary-image-03")
);
}
#[test]
fn summarize_entry_uses_requested_alias_for_cache_and_response_model_for_display() {
struct ResolvedModelProvider;
impl SummaryProvider for ResolvedModelProvider {
fn kind(&self) -> &'static str {
"anthropic_http"
}
fn model(&self) -> Option<&str> {
Some("claude-3-5-sonnet-latest")
}
fn summarize(&self, _: &SummaryRequest) -> Result<SummaryOutput> {
Ok(SummaryOutput {
text: r#"{"tldr":"t","summary":"s","tags":[]}"#.to_string(),
model: Some("claude-3-5-sonnet-20241022".to_string()),
})
}
}
let (_temp, paths, entry) = summary_image_fixture();
let record = summarize_entry(
&paths,
&entry.entry_uid,
SummaryBuildOptions::default(),
&ResolvedModelProvider,
"v1",
)
.unwrap();
assert_eq!(
record.provider_model.as_deref(),
Some("claude-3-5-sonnet-latest")
);
assert_eq!(
record.resolved_model.as_deref(),
Some("claude-3-5-sonnet-20241022")
);
} }
// ── Output normalization ─────────────────────────────────────────────── // ── Output normalization ───────────────────────────────────────────────
@ -1997,7 +2133,8 @@ mod tests {
#[test] #[test]
fn normalize_summary_json_recovers_json_wrapped_in_prose() { fn normalize_summary_json_recovers_json_wrapped_in_prose() {
let raw = "Sure! Here you go:\n{\"tldr\":\"t\",\"summary\":\"s\",\"tags\":[]}\nHope that helps."; let raw =
"Sure! Here you go:\n{\"tldr\":\"t\",\"summary\":\"s\",\"tags\":[]}\nHope that helps.";
let v: serde_json::Value = serde_json::from_str(&normalize_summary_json(raw)).unwrap(); let v: serde_json::Value = serde_json::from_str(&normalize_summary_json(raw)).unwrap();
assert_eq!(v["tldr"], "t"); assert_eq!(v["tldr"], "t");
} }
@ -2024,13 +2161,17 @@ mod tests {
#[test] #[test]
fn run_cli_kills_a_child_that_overruns_its_timeout() { fn run_cli_kills_a_child_that_overruns_its_timeout() {
let err = run_cli(Path::new("sleep"), &["30"], "", 1).unwrap_err().to_string(); let err = run_cli(Path::new("sleep"), &["30"], "", 1)
.unwrap_err()
.to_string();
assert!(err.contains("timed out"), "got: {err}"); assert!(err.contains("timed out"), "got: {err}");
} }
#[test] #[test]
fn run_cli_reports_a_nonzero_exit() { fn run_cli_reports_a_nonzero_exit() {
let err = run_cli(Path::new("false"), &[], "", 30).unwrap_err().to_string(); let err = run_cli(Path::new("false"), &[], "", 30)
.unwrap_err()
.to_string();
assert!(err.contains("exited with"), "got: {err}"); assert!(err.contains("exited with"), "got: {err}");
} }
} }

View file

@ -6,7 +6,7 @@
<title>Archivr</title> <title>Archivr</title>
<link rel="icon" type="image/svg+xml" href="/favicon.svg"> <link rel="icon" type="image/svg+xml" href="/favicon.svg">
<link rel="icon" type="image/x-icon" href="/favicon.ico"> <link rel="icon" type="image/x-icon" href="/favicon.ico">
<script type="module" crossorigin src="/assets/index-DH0j8cUl.js"></script> <script type="module" crossorigin src="/assets/index-DMKGhxnA.js"></script>
<link rel="stylesheet" crossorigin href="/assets/index-1h0SqIvL.css"> <link rel="stylesheet" crossorigin href="/assets/index-1h0SqIvL.css">
</head> </head>
<body> <body>

View file

@ -590,7 +590,9 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
)} )}
<p className="rail-summary-provider"> <p className="rail-summary-provider">
{PROVIDER_LABEL[summary.provider_kind] || summary.provider_kind} {PROVIDER_LABEL[summary.provider_kind] || summary.provider_kind}
{summary.provider_model ? ` \u00b7 ${summary.provider_model}` : ''} {summary.resolved_model || summary.provider_model
? ` \u00b7 ${summary.resolved_model || summary.provider_model}`
: ''}
</p> </p>
</div> </div>
)} )}