mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 12:55:00 +02:00
Merge branch 'sol-summary-core' into integration-all-three
# Conflicts: # crates/archivr-core/src/database.rs # crates/archivr-core/src/summarizer.rs
This commit is contained in:
commit
26df7dd2e9
6 changed files with 597 additions and 218 deletions
|
|
@ -951,7 +951,10 @@ fn register_tweet_artifacts(
|
|||
})?;
|
||||
for (role, raw_relpath) in tweet_raw_artifacts(&json_str)? {
|
||||
let raw_path = PathBuf::from(&raw_relpath);
|
||||
let blob = blob_record_for_raw_relpath(store_path, &raw_path)?;
|
||||
let mut blob = blob_record_for_raw_relpath(store_path, &raw_path)?;
|
||||
if role == "media" {
|
||||
blob.mime_type = tweet_media_image_mime(blob.extension.as_deref());
|
||||
}
|
||||
let blob_id = database::upsert_blob(conn, &blob)?;
|
||||
database::add_entry_artifact(
|
||||
conn,
|
||||
|
|
@ -1030,6 +1033,22 @@ fn record_tweet_entry(
|
|||
Ok(entry)
|
||||
}
|
||||
|
||||
/// Trusted image MIME types emitted by the X downloader's media paths.
|
||||
///
|
||||
/// Tweet JSON has no MIME field for ordinary downloaded media. Restricting this
|
||||
/// inference to the explicit image extensions keeps binary video/audio and
|
||||
/// unknown extensions out of multimodal summary input.
|
||||
fn tweet_media_image_mime(extension: Option<&str>) -> Option<String> {
|
||||
match extension?.to_ascii_lowercase().as_str() {
|
||||
"jpg" | "jpeg" => Some("image/jpeg".to_string()),
|
||||
"png" => Some("image/png".to_string()),
|
||||
"webp" => Some("image/webp".to_string()),
|
||||
"gif" => Some("image/gif".to_string()),
|
||||
"avif" => Some("image/avif".to_string()),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn tweet_raw_artifacts(tweet_json: &str) -> Result<Vec<(String, String)>> {
|
||||
let regex = regex::Regex::new(r#""(avatar_local_path|local_path)": "([^"\n]+)""#)?;
|
||||
let mut seen = HashSet::new();
|
||||
|
|
@ -1562,8 +1581,8 @@ pub fn perform_capture(
|
|||
let (rewritten, fonts) =
|
||||
downloader::font_extractor::extract_and_rewrite(&content, store_path, aid)
|
||||
.unwrap_or_else(|_| (content.clone(), vec![])); // non-fatal
|
||||
// Extract title after font-stripping so the title tag is not buried
|
||||
// behind multi-MB embedded font data that would exceed the 256 KiB window.
|
||||
// Extract title after font-stripping so the title tag is not buried
|
||||
// behind multi-MB embedded font data that would exceed the 256 KiB window.
|
||||
let title = downloader::singlefile::extract_html_title_str(&rewritten);
|
||||
fs::write(&temp_html, rewritten.as_bytes())
|
||||
.with_context(|| "failed to write rewritten HTML")?;
|
||||
|
|
@ -2810,7 +2829,11 @@ mod tests {
|
|||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
fs::write(
|
||||
archive_path.join("store_path"),
|
||||
store_path.to_str().unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path: archive_path.clone(),
|
||||
|
|
@ -2832,7 +2855,8 @@ mod tests {
|
|||
// Verify entry was created
|
||||
let conn = database::open_or_initialize(&archive_path).unwrap();
|
||||
let default_coll_id = database::ensure_default_collection(&conn).unwrap();
|
||||
let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
|
||||
let entries =
|
||||
archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
|
||||
assert_eq!(entries.len(), 1);
|
||||
let entry = &entries[0];
|
||||
assert_eq!(entry.title, Some(title.to_string()));
|
||||
|
|
@ -2859,7 +2883,11 @@ mod tests {
|
|||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
fs::write(
|
||||
archive_path.join("store_path"),
|
||||
store_path.to_str().unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path: archive_path.clone(),
|
||||
|
|
@ -2878,7 +2906,8 @@ mod tests {
|
|||
// Verify entry was created
|
||||
let conn = database::open_or_initialize(&archive_path).unwrap();
|
||||
let default_coll_id = database::ensure_default_collection(&conn).unwrap();
|
||||
let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
|
||||
let entries =
|
||||
archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
|
||||
assert_eq!(entries.len(), 1);
|
||||
let entry = &entries[0];
|
||||
assert_eq!(entry.title, Some(title.to_string()));
|
||||
|
|
@ -2901,7 +2930,11 @@ mod tests {
|
|||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
fs::write(
|
||||
archive_path.join("store_path"),
|
||||
store_path.to_str().unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path,
|
||||
|
|
@ -2911,7 +2944,10 @@ mod tests {
|
|||
|
||||
let result = perform_text_capture(&archive_paths, "", "Some body", "text/plain", None);
|
||||
assert!(result.is_err());
|
||||
assert!(result.unwrap_err().to_string().contains("title must not be empty"));
|
||||
assert!(result
|
||||
.unwrap_err()
|
||||
.to_string()
|
||||
.contains("title must not be empty"));
|
||||
|
||||
// Clean up
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
|
|
@ -2931,7 +2967,11 @@ mod tests {
|
|||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
fs::write(
|
||||
archive_path.join("store_path"),
|
||||
store_path.to_str().unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path,
|
||||
|
|
@ -2941,7 +2981,10 @@ mod tests {
|
|||
|
||||
let result = perform_text_capture(&archive_paths, "Some Title", "", "text/plain", None);
|
||||
assert!(result.is_err());
|
||||
assert!(result.unwrap_err().to_string().contains("body must not be empty"));
|
||||
assert!(result
|
||||
.unwrap_err()
|
||||
.to_string()
|
||||
.contains("body must not be empty"));
|
||||
|
||||
// Clean up
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
|
|
@ -2961,7 +3004,11 @@ mod tests {
|
|||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
fs::write(
|
||||
archive_path.join("store_path"),
|
||||
store_path.to_str().unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path,
|
||||
|
|
@ -2969,15 +3016,12 @@ mod tests {
|
|||
name: "test-archive".to_string(),
|
||||
};
|
||||
|
||||
let result = perform_text_capture(
|
||||
&archive_paths,
|
||||
"Title",
|
||||
"Body",
|
||||
"text/html",
|
||||
None,
|
||||
);
|
||||
let result = perform_text_capture(&archive_paths, "Title", "Body", "text/html", None);
|
||||
assert!(result.is_err());
|
||||
assert!(result.unwrap_err().to_string().contains("unsupported MIME type"));
|
||||
assert!(result
|
||||
.unwrap_err()
|
||||
.to_string()
|
||||
.contains("unsupported MIME type"));
|
||||
|
||||
// Clean up
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
|
|
@ -3003,12 +3047,15 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn test_record_tweet_entry_links_json_and_raw_artifacts() {
|
||||
let store_path = env::temp_dir().join(format!(
|
||||
"archivr-tweet-db-test-{}",
|
||||
Local::now().format("%Y%m%d%H%M%S%3f")
|
||||
));
|
||||
let _ = fs::remove_dir_all(&store_path);
|
||||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let archive_paths = archive::initialize_archive(
|
||||
temp.path(),
|
||||
&temp.path().join("store"),
|
||||
"Tweet summary test",
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
let store_path = &archive_paths.store_path;
|
||||
fs::create_dir_all(store_path.join("raw").join("a").join("b")).unwrap();
|
||||
fs::create_dir_all(store_path.join("raw").join("c").join("d")).unwrap();
|
||||
fs::write(
|
||||
|
|
@ -3025,21 +3072,21 @@ mod tests {
|
|||
.join("raw")
|
||||
.join("c")
|
||||
.join("d")
|
||||
.join("cdef01.mp4"),
|
||||
.join("cdef01.jpg"),
|
||||
b"media",
|
||||
)
|
||||
.unwrap();
|
||||
fs::write(
|
||||
store_path.join("raw_tweets").join("tweet-123.json"),
|
||||
r#"{
|
||||
"full_text": "Tweet body for summary selection.",
|
||||
"author": { "avatar_local_path": "raw/a/b/abcdef.jpg" },
|
||||
"entities": { "media": [{ "local_path": "raw/c/d/cdef01.mp4" }] }
|
||||
"entities": { "media": [{ "local_path": "raw/c/d/cdef01.jpg" }] }
|
||||
}"#,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let conn = rusqlite::Connection::open_in_memory().unwrap();
|
||||
database::initialize_schema(&conn).unwrap();
|
||||
let conn = database::open_or_initialize(&archive_paths.archive_path).unwrap();
|
||||
let user_id = database::ensure_default_user(&conn).unwrap();
|
||||
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
|
||||
let item = database::create_archive_run_item(
|
||||
|
|
@ -3088,10 +3135,26 @@ mod tests {
|
|||
|
||||
assert_eq!(artifact_count, 3);
|
||||
assert_eq!(blob_count, 2);
|
||||
let media_mime: Option<String> = conn
|
||||
.query_row(
|
||||
"SELECT b.mime_type FROM entry_artifacts ea JOIN blobs b ON b.id = ea.blob_id WHERE ea.entry_id = ?1 AND ea.artifact_role = 'media'",
|
||||
[entry.id],
|
||||
|row| row.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(media_mime.as_deref(), Some("image/jpeg"));
|
||||
let summary_input = crate::summarizer::build_summary_input(
|
||||
&archive_paths,
|
||||
&entry.entry_uid,
|
||||
crate::summarizer::SummaryBuildOptions {
|
||||
include_images: true,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(summary_input.request.images.len(), 1);
|
||||
assert_eq!(summary_input.request.images[0].mime_type, "image/jpeg");
|
||||
assert_eq!(run_status, "completed");
|
||||
assert!(store_path.join(&entry.structured_root_relpath).is_dir());
|
||||
|
||||
let _ = fs::remove_dir_all(store_path);
|
||||
}
|
||||
|
||||
mod title_tests {
|
||||
|
|
|
|||
|
|
@ -121,6 +121,8 @@ pub struct EntrySummaryRecord {
|
|||
pub summary_uid: String,
|
||||
pub entry_uid: String,
|
||||
pub provider_kind: String,
|
||||
/// Model selected by the provider after resolving the requested cache-key alias.
|
||||
pub resolved_model: Option<String>,
|
||||
pub provider_model: Option<String>,
|
||||
pub prompt_version: String,
|
||||
pub input_sha256: String,
|
||||
|
|
@ -350,6 +352,7 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
|
|||
entry_id INTEGER NOT NULL REFERENCES archived_entries(id) ON DELETE CASCADE,
|
||||
provider_kind TEXT NOT NULL,
|
||||
provider_model TEXT NOT NULL DEFAULT '',
|
||||
resolved_model TEXT,
|
||||
prompt_version TEXT NOT NULL,
|
||||
input_sha256 TEXT NOT NULL,
|
||||
status TEXT NOT NULL CHECK(status IN ('pending','running','completed','failed')),
|
||||
|
|
@ -477,6 +480,12 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
|
|||
// Migration: add notes_json column to existing capture_jobs tables.
|
||||
// Silently ignored when the column already exists (idempotent).
|
||||
let _ = conn.execute("ALTER TABLE capture_jobs ADD COLUMN notes_json TEXT", []);
|
||||
// Provider responses may resolve a requested alias to a concrete model.
|
||||
// Keep that display-only value outside the cache key.
|
||||
let _ = conn.execute(
|
||||
"ALTER TABLE entry_summaries ADD COLUMN resolved_model TEXT",
|
||||
[],
|
||||
);
|
||||
// Migration: add requires_auth column to existing collections tables.
|
||||
// Silently ignored when the column already exists (idempotent).
|
||||
let _ = conn.execute(
|
||||
|
|
@ -494,10 +503,11 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
|
|||
|row| row.get(0),
|
||||
)
|
||||
.optional()?;
|
||||
if summary_table_sql
|
||||
.as_deref()
|
||||
.is_some_and(|sql| sql.contains("UNIQUE(entry_id, provider_kind, provider_model, prompt_version, input_sha256)"))
|
||||
{
|
||||
if summary_table_sql.as_deref().is_some_and(|sql| {
|
||||
sql.contains(
|
||||
"UNIQUE(entry_id, provider_kind, provider_model, prompt_version, input_sha256)",
|
||||
)
|
||||
}) {
|
||||
conn.execute_batch(
|
||||
"BEGIN;
|
||||
CREATE TABLE entry_summaries_rebuilt (
|
||||
|
|
@ -506,6 +516,7 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
|
|||
entry_id INTEGER NOT NULL REFERENCES archived_entries(id) ON DELETE CASCADE,
|
||||
provider_kind TEXT NOT NULL,
|
||||
provider_model TEXT NOT NULL DEFAULT '',
|
||||
resolved_model TEXT,
|
||||
prompt_version TEXT NOT NULL,
|
||||
input_sha256 TEXT NOT NULL,
|
||||
status TEXT NOT NULL CHECK(status IN ('pending','running','completed','failed')),
|
||||
|
|
@ -517,7 +528,7 @@ pub fn initialize_schema(conn: &Connection) -> Result<()> {
|
|||
);
|
||||
INSERT INTO entry_summaries_rebuilt
|
||||
SELECT id, summary_uid, entry_id, provider_kind, COALESCE(provider_model, ''),
|
||||
prompt_version, input_sha256, status, summary_text, error_text,
|
||||
resolved_model, prompt_version, input_sha256, status, summary_text, error_text,
|
||||
created_at, updated_at, completed_at
|
||||
FROM entry_summaries;
|
||||
DROP TABLE entry_summaries;
|
||||
|
|
@ -1465,7 +1476,8 @@ pub fn fail_stalled_entry_summaries(conn: &Connection) -> Result<usize> {
|
|||
|
||||
/// `SELECT` list shared by every `entry_summaries` read, so all readers build
|
||||
/// an identical `EntrySummaryRecord` from the same column ordering.
|
||||
const ENTRY_SUMMARY_COLS: &str = "SELECT s.summary_uid, e.entry_uid, s.provider_kind, s.provider_model,
|
||||
const ENTRY_SUMMARY_COLS: &str =
|
||||
"SELECT s.summary_uid, e.entry_uid, s.provider_kind, s.provider_model, s.resolved_model,
|
||||
s.prompt_version, s.input_sha256, s.status, s.summary_text,
|
||||
s.error_text, s.created_at, s.updated_at, s.completed_at
|
||||
FROM entry_summaries s
|
||||
|
|
@ -1478,17 +1490,16 @@ fn map_entry_summary(row: &rusqlite::Row<'_>) -> rusqlite::Result<EntrySummaryRe
|
|||
provider_kind: row.get(2)?,
|
||||
// '' is the stored stand-in for "this provider has no model"; see the
|
||||
// doc comment on EntrySummaryRecord for why it is not NULL.
|
||||
provider_model: row
|
||||
.get::<_, Option<String>>(3)?
|
||||
.filter(|m| !m.is_empty()),
|
||||
prompt_version: row.get(4)?,
|
||||
input_sha256: row.get(5)?,
|
||||
status: row.get(6)?,
|
||||
summary_text: row.get(7)?,
|
||||
error_text: row.get(8)?,
|
||||
created_at: row.get(9)?,
|
||||
updated_at: row.get(10)?,
|
||||
completed_at: row.get(11)?,
|
||||
provider_model: row.get::<_, Option<String>>(3)?.filter(|m| !m.is_empty()),
|
||||
resolved_model: row.get(4)?,
|
||||
prompt_version: row.get(5)?,
|
||||
input_sha256: row.get(6)?,
|
||||
status: row.get(7)?,
|
||||
summary_text: row.get(8)?,
|
||||
error_text: row.get(9)?,
|
||||
created_at: row.get(10)?,
|
||||
updated_at: row.get(11)?,
|
||||
completed_at: row.get(12)?,
|
||||
})
|
||||
}
|
||||
|
||||
|
|
@ -1536,6 +1547,25 @@ pub fn upsert_pending_entry_summary(
|
|||
Ok(summary_uid)
|
||||
}
|
||||
|
||||
/// Completes a summary while retaining the provider's concrete response model
|
||||
/// for display. The requested provider model remains the cache-key identity.
|
||||
pub fn update_entry_summary_completed(
|
||||
conn: &Connection,
|
||||
summary_uid: &str,
|
||||
summary_text: &str,
|
||||
resolved_model: Option<&str>,
|
||||
) -> Result<()> {
|
||||
let now = now_timestamp();
|
||||
conn.execute(
|
||||
"UPDATE entry_summaries
|
||||
SET status = 'completed', summary_text = ?1, error_text = NULL,
|
||||
resolved_model = ?2, completed_at = ?3, updated_at = ?3
|
||||
WHERE summary_uid = ?4",
|
||||
rusqlite::params![summary_text, resolved_model, now, summary_uid],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Moves a summary row through `running` → `completed` / `failed`.
|
||||
/// `completed_at` is stamped only on the terminal `completed` transition.
|
||||
pub fn update_entry_summary_status(
|
||||
|
|
@ -1556,7 +1586,14 @@ pub fn update_entry_summary_status(
|
|||
SET status = ?1, summary_text = ?2, error_text = ?3,
|
||||
completed_at = ?4, updated_at = ?5
|
||||
WHERE summary_uid = ?6",
|
||||
rusqlite::params![status, summary_text, error_text, completed_at, now, summary_uid],
|
||||
rusqlite::params![
|
||||
status,
|
||||
summary_text,
|
||||
error_text,
|
||||
completed_at,
|
||||
now,
|
||||
summary_uid
|
||||
],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
|
@ -1733,7 +1770,11 @@ pub fn finish_archive_run(conn: &Connection, run_id: i64) -> Result<()> {
|
|||
[run_id],
|
||||
|row| row.get(0),
|
||||
)?;
|
||||
let status = if failed_count > 0 { "failed" } else { "completed" };
|
||||
let status = if failed_count > 0 {
|
||||
"failed"
|
||||
} else {
|
||||
"completed"
|
||||
};
|
||||
conn.execute(
|
||||
"UPDATE archive_runs SET status = ?1, finished_at = ?2 WHERE id = ?3",
|
||||
params![status, now_timestamp(), run_id],
|
||||
|
|
@ -1969,9 +2010,8 @@ pub fn delete_entry(conn: &Connection, entry_uid: &str) -> Result<bool> {
|
|||
// (no grandchildren), so without `id = ?1` the set would be empty and
|
||||
// cascade_cached_bytes_after_subtree_delete would not recalculate shared-blob totals.
|
||||
let subtree_ids: Vec<i64> = {
|
||||
let mut stmt = conn.prepare(
|
||||
"SELECT id FROM archived_entries WHERE id = ?1 OR root_entry_id = ?1",
|
||||
)?;
|
||||
let mut stmt =
|
||||
conn.prepare("SELECT id FROM archived_entries WHERE id = ?1 OR root_entry_id = ?1")?;
|
||||
stmt.query_map([entry_id], |row| row.get(0))?
|
||||
.collect::<rusqlite::Result<_>>()?
|
||||
};
|
||||
|
|
@ -2622,7 +2662,6 @@ pub fn visibility_to_bits(visibility: &str) -> u32 {
|
|||
}
|
||||
}
|
||||
|
||||
|
||||
/// Returns the id of the '_default_' collection, creating it if absent.
|
||||
pub fn ensure_default_collection(conn: &Connection) -> Result<i64> {
|
||||
let now = now_timestamp();
|
||||
|
|
@ -2725,16 +2764,20 @@ pub fn get_collection_by_slug(conn: &Connection, slug: &str) -> Result<Option<Co
|
|||
"SELECT id, collection_uid, name, slug, default_visibility_bits, created_at, requires_auth \
|
||||
FROM collections WHERE slug = ?1",
|
||||
[slug],
|
||||
|row| Ok(CollectionRecord {
|
||||
id: row.get(0)?,
|
||||
collection_uid: row.get(1)?,
|
||||
name: row.get(2)?,
|
||||
slug: row.get(3)?,
|
||||
default_visibility_bits: row.get::<_, i64>(4)? as u32,
|
||||
created_at: row.get(5)?,
|
||||
requires_auth: row.get::<_, i64>(6)? != 0,
|
||||
}),
|
||||
).optional().map_err(Into::into)
|
||||
|row| {
|
||||
Ok(CollectionRecord {
|
||||
id: row.get(0)?,
|
||||
collection_uid: row.get(1)?,
|
||||
name: row.get(2)?,
|
||||
slug: row.get(3)?,
|
||||
default_visibility_bits: row.get::<_, i64>(4)? as u32,
|
||||
created_at: row.get(5)?,
|
||||
requires_auth: row.get::<_, i64>(6)? != 0,
|
||||
})
|
||||
},
|
||||
)
|
||||
.optional()
|
||||
.map_err(Into::into)
|
||||
}
|
||||
|
||||
/// Adds an entry to a collection with given visibility_bits. Idempotent (INSERT OR IGNORE).
|
||||
|
|
@ -2793,7 +2836,12 @@ pub fn get_entry_collection_memberships(
|
|||
WHERE ce.entry_id = ?1",
|
||||
)?;
|
||||
stmt.query_map([entry_id], |row| {
|
||||
Ok((row.get(0)?, row.get(1)?, row.get(2)?, row.get::<_, i64>(3)? as u32))
|
||||
Ok((
|
||||
row.get(0)?,
|
||||
row.get(1)?,
|
||||
row.get(2)?,
|
||||
row.get::<_, i64>(3)? as u32,
|
||||
))
|
||||
})?
|
||||
.collect::<Result<_, _>>()
|
||||
.map_err(Into::into)
|
||||
|
|
@ -3351,7 +3399,10 @@ mod tests {
|
|||
|row| row.get::<_, i64>(0).map(|v| v as u32),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(default_bits, 2, "default collection should start USER-visible");
|
||||
assert_eq!(
|
||||
default_bits, 2,
|
||||
"default collection should start USER-visible"
|
||||
);
|
||||
|
||||
// Create an entry with visibility = "private" (what capture always passes).
|
||||
let entry = create_entry_fixture(&conn, "private", None, None);
|
||||
|
|
@ -3399,7 +3450,10 @@ mod tests {
|
|||
|row| row.get::<_, i64>(0).map(|v| v as u32),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(public_bits, 3, "collection default=public should produce bits=3");
|
||||
assert_eq!(
|
||||
public_bits, 3,
|
||||
"collection default=public should produce bits=3"
|
||||
);
|
||||
|
||||
// Child entries must NOT use the collection default — they keep
|
||||
// visibility_to_bits(entry.visibility) so parent-child visibility
|
||||
|
|
@ -3412,7 +3466,10 @@ mod tests {
|
|||
|row| row.get::<_, i64>(0).map(|v| v as u32),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(child_bits, 0, "child entries must use visibility_to_bits, not collection default");
|
||||
assert_eq!(
|
||||
child_bits, 0,
|
||||
"child entries must use visibility_to_bits, not collection default"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
|
@ -4441,30 +4498,50 @@ mod tests {
|
|||
|
||||
// Root item (parent_item_id IS NULL) — mirrors what record_container_entry does.
|
||||
let root_item = create_archive_run_item(
|
||||
&c, run.id, None, 0, "https://example.com/pl", None, "youtube", "playlist",
|
||||
).unwrap();
|
||||
&c,
|
||||
run.id,
|
||||
None,
|
||||
0,
|
||||
"https://example.com/pl",
|
||||
None,
|
||||
"youtube",
|
||||
"playlist",
|
||||
)
|
||||
.unwrap();
|
||||
// Child item (parent_item_id IS NOT NULL).
|
||||
let child_item = create_archive_run_item(
|
||||
&c, run.id, Some(root_item.id), 1, "https://example.com/v1", None, "youtube", "video",
|
||||
).unwrap();
|
||||
&c,
|
||||
run.id,
|
||||
Some(root_item.id),
|
||||
1,
|
||||
"https://example.com/v1",
|
||||
None,
|
||||
"youtube",
|
||||
"video",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Complete both — marks archive_runs.completed_count = 2.
|
||||
c.execute(
|
||||
"UPDATE archive_run_items SET status = 'completed' WHERE id IN (?1, ?2)",
|
||||
rusqlite::params![root_item.id, child_item.id],
|
||||
).unwrap();
|
||||
)
|
||||
.unwrap();
|
||||
refresh_run_counters(&c, run.id).unwrap();
|
||||
|
||||
let total: i64 = c.query_row(
|
||||
"SELECT completed_count FROM archive_runs WHERE id = ?1", [run.id], |r| r.get(0),
|
||||
).unwrap();
|
||||
let total: i64 = c
|
||||
.query_row(
|
||||
"SELECT completed_count FROM archive_runs WHERE id = ?1",
|
||||
[run.id],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(total, 2, "both items completed: DB counter must be 2");
|
||||
|
||||
let child_count = get_run_completed_child_count(&c, run.id).unwrap();
|
||||
assert_eq!(child_count, 1, "only the child item must be counted");
|
||||
}
|
||||
|
||||
|
||||
// ── entry_summaries ────────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
|
|
@ -4489,6 +4566,41 @@ mod tests {
|
|||
assert!(latest_entry_summary(&c, entry.id).unwrap().is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn initialize_schema_migrates_resolved_model_for_existing_summary_table() {
|
||||
let c = Connection::open_in_memory().unwrap();
|
||||
c.execute_batch(
|
||||
"CREATE TABLE entry_summaries (
|
||||
id INTEGER PRIMARY KEY,
|
||||
summary_uid TEXT NOT NULL UNIQUE,
|
||||
entry_id INTEGER NOT NULL,
|
||||
provider_kind TEXT NOT NULL,
|
||||
provider_model TEXT,
|
||||
prompt_version TEXT NOT NULL,
|
||||
input_sha256 TEXT NOT NULL,
|
||||
status TEXT NOT NULL,
|
||||
summary_text TEXT,
|
||||
error_text TEXT,
|
||||
created_at TEXT NOT NULL,
|
||||
updated_at TEXT NOT NULL,
|
||||
completed_at TEXT,
|
||||
UNIQUE(entry_id, provider_kind, provider_model, prompt_version, input_sha256)
|
||||
);",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
initialize_schema(&c).unwrap();
|
||||
|
||||
let has_resolved_model: i64 = c
|
||||
.query_row(
|
||||
"SELECT COUNT(*) FROM pragma_table_info('entry_summaries') WHERE name = 'resolved_model'",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(has_resolved_model, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entry_summary_lifecycle_pending_running_completed() {
|
||||
let c = conn();
|
||||
|
|
@ -4519,6 +4631,48 @@ mod tests {
|
|||
assert!(rec.completed_at.is_some(), "completed rows must be stamped");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entry_summary_keeps_requested_model_for_cache_and_resolved_model_for_display() {
|
||||
let c = conn();
|
||||
let entry = create_entry_fixture(&c, "private", None, None);
|
||||
let uid = upsert_pending_entry_summary(
|
||||
&c,
|
||||
entry.id,
|
||||
"anthropic_http",
|
||||
Some("claude-3-5-sonnet-latest"),
|
||||
"v1",
|
||||
"sha1",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
update_entry_summary_completed(&c, &uid, "summary", Some("claude-3-5-sonnet-20241022"))
|
||||
.unwrap();
|
||||
let rec = get_entry_summary_by_uid(&c, &uid).unwrap().unwrap();
|
||||
|
||||
assert_eq!(
|
||||
rec.provider_model.as_deref(),
|
||||
Some("claude-3-5-sonnet-latest")
|
||||
);
|
||||
assert_eq!(
|
||||
rec.resolved_model.as_deref(),
|
||||
Some("claude-3-5-sonnet-20241022")
|
||||
);
|
||||
assert_eq!(
|
||||
find_entry_summary(
|
||||
&c,
|
||||
entry.id,
|
||||
"anthropic_http",
|
||||
Some("claude-3-5-sonnet-latest"),
|
||||
"v1",
|
||||
"sha1",
|
||||
)
|
||||
.unwrap()
|
||||
.unwrap()
|
||||
.summary_uid,
|
||||
uid,
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entry_summary_failure_records_error_and_no_completed_at() {
|
||||
let c = conn();
|
||||
|
|
@ -4569,12 +4723,12 @@ mod tests {
|
|||
fn fail_stalled_entry_summaries_marks_pending_and_running_with_restart_message() {
|
||||
let c = conn();
|
||||
let entry = create_entry_fixture(&c, "private", None, None);
|
||||
let pending = upsert_pending_entry_summary(&c, entry.id, "claude_cli", None, "v1", "a")
|
||||
.unwrap();
|
||||
let running = upsert_pending_entry_summary(&c, entry.id, "codex_cli", None, "v1", "b")
|
||||
.unwrap();
|
||||
let completed = upsert_pending_entry_summary(&c, entry.id, "anthropic_http", None, "v1", "c")
|
||||
.unwrap();
|
||||
let pending =
|
||||
upsert_pending_entry_summary(&c, entry.id, "claude_cli", None, "v1", "a").unwrap();
|
||||
let running =
|
||||
upsert_pending_entry_summary(&c, entry.id, "codex_cli", None, "v1", "b").unwrap();
|
||||
let completed =
|
||||
upsert_pending_entry_summary(&c, entry.id, "anthropic_http", None, "v1", "c").unwrap();
|
||||
update_entry_summary_status(&c, &running, "running", None, None).unwrap();
|
||||
update_entry_summary_status(&c, &completed, "completed", Some("done"), None).unwrap();
|
||||
|
||||
|
|
@ -4589,7 +4743,10 @@ mod tests {
|
|||
assert!(!row.updated_at.is_empty());
|
||||
}
|
||||
assert_eq!(
|
||||
get_entry_summary_by_uid(&c, &completed).unwrap().unwrap().status,
|
||||
get_entry_summary_by_uid(&c, &completed)
|
||||
.unwrap()
|
||||
.unwrap()
|
||||
.status,
|
||||
"completed"
|
||||
);
|
||||
}
|
||||
|
|
@ -4637,9 +4794,16 @@ mod tests {
|
|||
.unwrap();
|
||||
assert_eq!(hit.summary_uid, uid);
|
||||
assert!(
|
||||
find_entry_summary(&c, entry.id, "openai_compatible", Some("gpt"), "v1", "changed")
|
||||
.unwrap()
|
||||
.is_none(),
|
||||
find_entry_summary(
|
||||
&c,
|
||||
entry.id,
|
||||
"openai_compatible",
|
||||
Some("gpt"),
|
||||
"v1",
|
||||
"changed"
|
||||
)
|
||||
.unwrap()
|
||||
.is_none(),
|
||||
"a changed input digest must miss the cache"
|
||||
);
|
||||
}
|
||||
|
|
@ -4653,7 +4817,13 @@ mod tests {
|
|||
update_entry_summary_status(&c, &a, "completed", Some("first"), None).unwrap();
|
||||
update_entry_summary_status(&c, &b, "completed", Some("second"), None).unwrap();
|
||||
// Same-timestamp ties break on id DESC, so the later insert wins either way.
|
||||
assert_eq!(latest_entry_summary(&c, entry.id).unwrap().unwrap().summary_uid, b);
|
||||
assert_eq!(
|
||||
latest_entry_summary(&c, entry.id)
|
||||
.unwrap()
|
||||
.unwrap()
|
||||
.summary_uid,
|
||||
b
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
|
@ -4680,7 +4850,10 @@ mod tests {
|
|||
fn entry_id_for_uid_resolves_and_misses() {
|
||||
let c = conn();
|
||||
let entry = create_entry_fixture(&c, "private", None, None);
|
||||
assert_eq!(entry_id_for_uid(&c, &entry.entry_uid).unwrap(), Some(entry.id));
|
||||
assert_eq!(
|
||||
entry_id_for_uid(&c, &entry.entry_uid).unwrap(),
|
||||
Some(entry.id)
|
||||
);
|
||||
assert_eq!(entry_id_for_uid(&c, "ent_nope").unwrap(), None);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -253,13 +253,19 @@ fn resolve_cli(env_name: &str, well_known_absolute: &[&str], bare: &str) -> Path
|
|||
pub fn provider_from_env(kind: &str) -> Result<ProviderConfig> {
|
||||
match kind {
|
||||
"anthropic_http" => Ok(ProviderConfig::AnthropicHttp(HttpProviderConfig {
|
||||
endpoint: env_or("ARCHIVR_ANTHROPIC_URL", "https://api.anthropic.com/v1/messages"),
|
||||
endpoint: env_or(
|
||||
"ARCHIVR_ANTHROPIC_URL",
|
||||
"https://api.anthropic.com/v1/messages",
|
||||
),
|
||||
api_key: required_env("ARCHIVR_ANTHROPIC_API_KEY")?,
|
||||
model: env_or("ARCHIVR_ANTHROPIC_MODEL", "claude-3-5-sonnet-latest"),
|
||||
timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS),
|
||||
})),
|
||||
"openai_compatible" => Ok(ProviderConfig::OpenAiCompatible(HttpProviderConfig {
|
||||
endpoint: env_or("ARCHIVR_OPENAI_URL", "https://api.openai.com/v1/chat/completions"),
|
||||
endpoint: env_or(
|
||||
"ARCHIVR_OPENAI_URL",
|
||||
"https://api.openai.com/v1/chat/completions",
|
||||
),
|
||||
api_key: required_env("ARCHIVR_OPENAI_API_KEY")?,
|
||||
model: env_or("ARCHIVR_OPENAI_MODEL", "gpt-4o-mini"),
|
||||
timeout_secs: env_timeout("ARCHIVR_SUMMARY_HTTP_TIMEOUT", DEFAULT_HTTP_TIMEOUT_SECS),
|
||||
|
|
@ -328,8 +334,12 @@ fn http_client(timeout_secs: u64) -> Result<reqwest::blocking::Client> {
|
|||
/// Body builder kept separate from the transport so it can be unit-tested
|
||||
/// without a network round-trip.
|
||||
fn read_image_base64(image: &SummaryImage) -> Result<String> {
|
||||
let bytes = std::fs::read(&image.archive_file)
|
||||
.with_context(|| format!("failed to read summary image {}", image.archive_file.display()))?;
|
||||
let bytes = std::fs::read(&image.archive_file).with_context(|| {
|
||||
format!(
|
||||
"failed to read summary image {}",
|
||||
image.archive_file.display()
|
||||
)
|
||||
})?;
|
||||
Ok(base64::engine::general_purpose::STANDARD.encode(bytes))
|
||||
}
|
||||
|
||||
|
|
@ -420,7 +430,10 @@ impl SummaryProvider for AnthropicHttpProvider {
|
|||
let status = resp.status();
|
||||
let text = resp.text().unwrap_or_default();
|
||||
if !status.is_success() {
|
||||
bail!("anthropic API returned {status}: {}", truncate_for_error(&text));
|
||||
bail!(
|
||||
"anthropic API returned {status}: {}",
|
||||
truncate_for_error(&text)
|
||||
);
|
||||
}
|
||||
parse_anthropic_response(&text)
|
||||
}
|
||||
|
|
@ -499,12 +512,7 @@ fn truncate_for_error(s: &str) -> String {
|
|||
/// the child if it overruns. stdin is written on a third thread because a
|
||||
/// 48 KB prompt can exceed the pipe buffer, and writing it inline would
|
||||
/// deadlock against a child that is waiting for us to read its output.
|
||||
fn run_cli(
|
||||
executable: &Path,
|
||||
args: &[&str],
|
||||
prompt: &str,
|
||||
timeout_secs: u64,
|
||||
) -> Result<String> {
|
||||
fn run_cli(executable: &Path, args: &[&str], prompt: &str, timeout_secs: u64) -> Result<String> {
|
||||
let mut child = Command::new(executable)
|
||||
.args(args)
|
||||
.stdin(Stdio::piped())
|
||||
|
|
@ -538,14 +546,13 @@ fn run_cli(
|
|||
});
|
||||
|
||||
let collected = match rx.recv_timeout(Duration::from_secs(timeout_secs)) {
|
||||
Ok(res) => res.with_context(|| format!("failed to read stdout of {}", executable.display()))?,
|
||||
Ok(res) => {
|
||||
res.with_context(|| format!("failed to read stdout of {}", executable.display()))?
|
||||
}
|
||||
Err(_) => {
|
||||
let _ = child.kill();
|
||||
let _ = child.wait();
|
||||
bail!(
|
||||
"{} timed out after {timeout_secs}s",
|
||||
executable.display()
|
||||
);
|
||||
bail!("{} timed out after {timeout_secs}s", executable.display());
|
||||
}
|
||||
};
|
||||
|
||||
|
|
@ -588,7 +595,12 @@ impl SummaryProvider for ClaudeCliProvider {
|
|||
args.push("--model");
|
||||
args.push(model);
|
||||
}
|
||||
let out = run_cli(&self.0.executable, &args, &build_combined_prompt(request), self.0.timeout_secs)?;
|
||||
let out = run_cli(
|
||||
&self.0.executable,
|
||||
&args,
|
||||
&build_combined_prompt(request),
|
||||
self.0.timeout_secs,
|
||||
)?;
|
||||
Ok(SummaryOutput {
|
||||
text: out,
|
||||
model: self.0.model.clone(),
|
||||
|
|
@ -638,7 +650,11 @@ mod codex {
|
|||
let out = std::fs::read_to_string(path).ok()?;
|
||||
let _ = std::fs::remove_file(path);
|
||||
let trimmed = out.trim();
|
||||
if trimmed.is_empty() { None } else { Some(trimmed.to_string()) }
|
||||
if trimmed.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(trimmed.to_string())
|
||||
}
|
||||
}
|
||||
|
||||
pub fn primary_args(
|
||||
|
|
@ -698,30 +714,21 @@ mod codex {
|
|||
|
||||
// Fallback: positional prompt, no stdin, same --output-last-message.
|
||||
let fb = positional_args(images, &out_path, cfg.model.as_deref(), prompt);
|
||||
let out = Command::new(&cfg.executable)
|
||||
.args(&fb)
|
||||
.output()
|
||||
.with_context(|| {
|
||||
format!(
|
||||
"codex `exec -` failed ({primary_err:#}); positional fallback also failed to spawn{}",
|
||||
missing_binary_hint(cfg)
|
||||
)
|
||||
})?;
|
||||
if !out.status.success() {
|
||||
let fb_refs: Vec<&str> = fb.iter().map(String::as_str).collect();
|
||||
let out = run_cli(&cfg.executable, &fb_refs, "", cfg.timeout_secs).with_context(|| {
|
||||
let _ = std::fs::remove_file(&out_path);
|
||||
bail!(
|
||||
"codex `exec -` failed ({primary_err:#}); positional fallback exited with {}: {}",
|
||||
out.status,
|
||||
truncate_for_error(&String::from_utf8_lossy(&out.stderr))
|
||||
);
|
||||
}
|
||||
format!(
|
||||
"codex `exec -` failed ({primary_err:#}); positional fallback failed{}",
|
||||
missing_binary_hint(cfg)
|
||||
)
|
||||
})?;
|
||||
if let Some(text) = read_and_cleanup(&out_path) {
|
||||
return Ok(text);
|
||||
}
|
||||
// Last resort — the child succeeded but wrote nothing to the file. Fall
|
||||
// back to raw stdout so the caller has *something* to normalize.
|
||||
let _ = std::fs::remove_file(&out_path);
|
||||
Ok(String::from_utf8_lossy(&out.stdout).to_string())
|
||||
Ok(out)
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -771,7 +778,8 @@ pub fn strip_html(html: &str) -> String {
|
|||
.unwrap();
|
||||
let comments = regex::Regex::new(r"(?s)<!--.*?-->").unwrap();
|
||||
// Treat block-level closers as line breaks so paragraphs do not run together.
|
||||
let breaks = regex::Regex::new(r"(?i)</\s*(p|div|br|li|h[1-6]|tr|section|article)\s*>").unwrap();
|
||||
let breaks =
|
||||
regex::Regex::new(r"(?i)</\s*(p|div|br|li|h[1-6]|tr|section|article)\s*>").unwrap();
|
||||
let tags = regex::Regex::new(r"(?s)<[^>]*>").unwrap();
|
||||
let spaces = regex::Regex::new(r"[ \t\r\f\v]+").unwrap();
|
||||
let blank_lines = regex::Regex::new(r"\n{3,}").unwrap();
|
||||
|
|
@ -782,11 +790,7 @@ pub fn strip_html(html: &str) -> String {
|
|||
let s = tags.replace_all(&s, " ");
|
||||
let s = decode_entities(&s);
|
||||
let s = spaces.replace_all(&s, " ");
|
||||
let s = s
|
||||
.lines()
|
||||
.map(str::trim)
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n");
|
||||
let s = s.lines().map(str::trim).collect::<Vec<_>>().join("\n");
|
||||
blank_lines.replace_all(&s, "\n\n").trim().to_string()
|
||||
}
|
||||
|
||||
|
|
@ -948,9 +952,12 @@ fn load_summary_image_candidates(
|
|||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
) -> Result<Vec<SummaryImage>> {
|
||||
let canonical_store = store_path
|
||||
.canonicalize()
|
||||
.with_context(|| format!("failed to canonicalize store path: {}", store_path.display()))?;
|
||||
let canonical_store = store_path.canonicalize().with_context(|| {
|
||||
format!(
|
||||
"failed to canonicalize store path: {}",
|
||||
store_path.display()
|
||||
)
|
||||
})?;
|
||||
let mut stmt = conn.prepare(
|
||||
"SELECT b.sha256, b.mime_type, b.extension, b.byte_size, ea.relpath
|
||||
FROM entry_artifacts ea
|
||||
|
|
@ -1050,7 +1057,11 @@ pub fn build_summary_input(
|
|||
// Load every matching artifact in insertion order so a thread summarizes
|
||||
// as the whole conversation, not just its first status.
|
||||
let is_tweetish = matches!(entity_kind.as_str(), "tweet" | "tweet_thread");
|
||||
let primary_role = if is_tweetish { "raw_tweet_json" } else { "primary_media" };
|
||||
let primary_role = if is_tweetish {
|
||||
"raw_tweet_json"
|
||||
} else {
|
||||
"primary_media"
|
||||
};
|
||||
let mut artifacts = load_summary_artifacts(&conn, entry_id, primary_role)?;
|
||||
if artifacts.is_empty() && is_tweetish {
|
||||
// Older archives may have stored tweet payloads under `primary_media`.
|
||||
|
|
@ -1066,8 +1077,11 @@ pub fn build_summary_input(
|
|||
let ext = extension_of(relpath);
|
||||
let mime = mime_opt.clone().unwrap_or_default();
|
||||
|
||||
let piece = if ext == "md" || ext == "markdown" || ext == "txt"
|
||||
|| mime.starts_with("text/markdown") || mime == "text/plain"
|
||||
let piece = if ext == "md"
|
||||
|| ext == "markdown"
|
||||
|| ext == "txt"
|
||||
|| mime.starts_with("text/markdown")
|
||||
|| mime == "text/plain"
|
||||
{
|
||||
std::fs::read_to_string(&abs)
|
||||
.with_context(|| format!("failed to read {}", abs.display()))?
|
||||
|
|
@ -1162,8 +1176,9 @@ pub fn normalize_summary_json(raw: &str) -> String {
|
|||
/// Owns the whole row lifecycle (`pending` → `running` → `completed`/`failed`)
|
||||
/// so a caller running it on a background thread only has to handle the
|
||||
/// `Err` case. The `(entry_id, provider_kind, provider_model, prompt_version,
|
||||
/// input_sha256)` UNIQUE key means re-running against unchanged input reuses the
|
||||
/// same row rather than accumulating duplicates.
|
||||
/// input_sha256)` cache key identifies equivalent requests. Each generation is
|
||||
/// nevertheless recorded as a distinct attempt, so a forced regeneration cannot
|
||||
/// hide a prior completed result while the new attempt is pending or running.
|
||||
pub fn summarize_entry(
|
||||
archive_paths: &ArchivePaths,
|
||||
entry_uid: &str,
|
||||
|
|
@ -1184,12 +1199,7 @@ pub fn summarize_entry(
|
|||
prompt_version,
|
||||
&input.input_sha256,
|
||||
)?;
|
||||
summarize_prebuilt_entry(
|
||||
archive_paths,
|
||||
input,
|
||||
&summary_uid,
|
||||
provider,
|
||||
)
|
||||
summarize_prebuilt_entry(archive_paths, input, &summary_uid, provider)
|
||||
}
|
||||
|
||||
/// Runs a previously validated and claimed summary attempt.
|
||||
|
|
@ -1209,23 +1219,16 @@ pub fn summarize_prebuilt_entry(
|
|||
match provider.summarize(&input.request) {
|
||||
Ok(output) => {
|
||||
let text = normalize_summary_json(&output.text);
|
||||
database::update_entry_summary_status(
|
||||
database::update_entry_summary_completed(
|
||||
&conn,
|
||||
summary_uid,
|
||||
"completed",
|
||||
Some(&text),
|
||||
None,
|
||||
&text,
|
||||
output.model.as_deref(),
|
||||
)?;
|
||||
}
|
||||
Err(e) => {
|
||||
let msg = format!("{e:#}");
|
||||
database::update_entry_summary_status(
|
||||
&conn,
|
||||
summary_uid,
|
||||
"failed",
|
||||
None,
|
||||
Some(&msg),
|
||||
)?;
|
||||
database::update_entry_summary_status(&conn, summary_uid, "failed", None, Some(&msg))?;
|
||||
return Err(e);
|
||||
}
|
||||
}
|
||||
|
|
@ -1317,7 +1320,9 @@ mod tests {
|
|||
fn provider_from_env_openai_missing_key_names_the_variable() {
|
||||
let _g = ENV_LOCK.lock().unwrap();
|
||||
clear_provider_env();
|
||||
let err = provider_from_env("openai_compatible").unwrap_err().to_string();
|
||||
let err = provider_from_env("openai_compatible")
|
||||
.unwrap_err()
|
||||
.to_string();
|
||||
assert!(err.contains("ARCHIVR_OPENAI_API_KEY"), "got: {err}");
|
||||
}
|
||||
|
||||
|
|
@ -1327,7 +1332,10 @@ mod tests {
|
|||
clear_provider_env();
|
||||
unsafe {
|
||||
env::set_var("ARCHIVR_OPENAI_API_KEY", "k");
|
||||
env::set_var("ARCHIVR_OPENAI_URL", "http://localhost:1234/v1/chat/completions");
|
||||
env::set_var(
|
||||
"ARCHIVR_OPENAI_URL",
|
||||
"http://localhost:1234/v1/chat/completions",
|
||||
);
|
||||
env::set_var("ARCHIVR_OPENAI_MODEL", "local-model");
|
||||
env::set_var("ARCHIVR_SUMMARY_HTTP_TIMEOUT", "7");
|
||||
}
|
||||
|
|
@ -1353,7 +1361,10 @@ mod tests {
|
|||
// file-name-`claude` path.
|
||||
assert!(
|
||||
c.executable == PathBuf::from("claude")
|
||||
|| c.executable.file_name().map(|f| f == "claude").unwrap_or(false),
|
||||
|| c.executable
|
||||
.file_name()
|
||||
.map(|f| f == "claude")
|
||||
.unwrap_or(false),
|
||||
"unexpected claude executable: {}",
|
||||
c.executable.display()
|
||||
);
|
||||
|
|
@ -1368,7 +1379,10 @@ mod tests {
|
|||
// deliberately opportunistic.
|
||||
assert!(
|
||||
c.executable == PathBuf::from("codex")
|
||||
|| c.executable.file_name().map(|f| f == "codex").unwrap_or(false),
|
||||
|| c.executable
|
||||
.file_name()
|
||||
.map(|f| f == "codex")
|
||||
.unwrap_or(false),
|
||||
"unexpected codex executable: {}",
|
||||
c.executable.display()
|
||||
);
|
||||
|
|
@ -1479,29 +1493,57 @@ mod tests {
|
|||
.position(|arg| arg == "--output-last-message")
|
||||
.unwrap();
|
||||
assert!(image_at < output_at);
|
||||
assert_eq!(args[image_at + 1], request.images[0].archive_file.to_string_lossy());
|
||||
assert_eq!(
|
||||
args[image_at + 1],
|
||||
request.images[0].archive_file.to_string_lossy()
|
||||
);
|
||||
assert_eq!(args.last().unwrap(), "-");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn codex_positional_arguments_put_images_before_output_path() {
|
||||
let (_temp, request) = image_request();
|
||||
let args = codex::positional_args(
|
||||
&request.images,
|
||||
Path::new("/tmp/output"),
|
||||
None,
|
||||
"prompt",
|
||||
);
|
||||
let args =
|
||||
codex::positional_args(&request.images, Path::new("/tmp/output"), None, "prompt");
|
||||
let image_at = args.iter().position(|arg| arg == "--image").unwrap();
|
||||
let output_at = args
|
||||
.iter()
|
||||
.position(|arg| arg == "--output-last-message")
|
||||
.unwrap();
|
||||
assert!(image_at < output_at);
|
||||
assert_eq!(args[image_at + 1], request.images[0].archive_file.to_string_lossy());
|
||||
assert_eq!(
|
||||
args[image_at + 1],
|
||||
request.images[0].archive_file.to_string_lossy()
|
||||
);
|
||||
assert_eq!(args.last().unwrap(), "prompt");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn codex_positional_fallback_honors_cli_timeout() {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let executable = temp.path().join("codex-fixture.sh");
|
||||
std::fs::write(
|
||||
&executable,
|
||||
"#!/bin/sh\nfor arg in \"$@\"; do [ \"$arg\" = \"-\" ] && exit 1; done\nsleep 30\n",
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::set_permissions(&executable, std::fs::Permissions::from_mode(0o755)).unwrap();
|
||||
let cfg = CliProviderConfig {
|
||||
executable,
|
||||
model: None,
|
||||
timeout_secs: 1,
|
||||
};
|
||||
|
||||
let started = std::time::Instant::now();
|
||||
let err = format!("{:#}", codex::run(&cfg, "prompt", &[]).unwrap_err());
|
||||
|
||||
assert!(started.elapsed() < Duration::from_secs(5));
|
||||
assert!(err.contains("timed out after 1s"), "got: {err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn claude_rejects_images_before_spawning() {
|
||||
let (_temp, request) = image_request();
|
||||
|
|
@ -1511,12 +1553,16 @@ mod tests {
|
|||
timeout_secs: 1,
|
||||
});
|
||||
let err = provider.summarize(&request).unwrap_err().to_string();
|
||||
assert!(err.contains("Claude CLI cannot attach local images"), "got: {err}");
|
||||
assert!(
|
||||
err.contains("Claude CLI cannot attach local images"),
|
||||
"got: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_anthropic_response_extracts_text_and_model() {
|
||||
let body = r#"{"model":"claude-3-5-sonnet-20241022","content":[{"type":"text","text":"hi"}]}"#;
|
||||
let body =
|
||||
r#"{"model":"claude-3-5-sonnet-20241022","content":[{"type":"text","text":"hi"}]}"#;
|
||||
let out = parse_anthropic_response(body).unwrap();
|
||||
assert_eq!(out.text, "hi");
|
||||
assert_eq!(out.model.as_deref(), Some("claude-3-5-sonnet-20241022"));
|
||||
|
|
@ -1559,7 +1605,10 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn strip_html_decodes_common_entities() {
|
||||
assert_eq!(strip_html("<p>a & b c</p>").replace('\u{a0}', " "), "a & b c");
|
||||
assert_eq!(
|
||||
strip_html("<p>a & b c</p>").replace('\u{a0}', " "),
|
||||
"a & b c"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
|
@ -1727,8 +1776,14 @@ mod tests {
|
|||
.unwrap();
|
||||
}
|
||||
|
||||
let input = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()).unwrap();
|
||||
assert!(input.request.content.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body."));
|
||||
let input =
|
||||
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default()).unwrap();
|
||||
assert!(
|
||||
input
|
||||
.request
|
||||
.content
|
||||
.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body.")
|
||||
);
|
||||
}
|
||||
|
||||
fn summary_image_fixture() -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
|
||||
|
|
@ -1800,19 +1855,12 @@ mod tests {
|
|||
)
|
||||
.unwrap();
|
||||
|
||||
let no_artifact = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
|
||||
.unwrap_err();
|
||||
let no_artifact =
|
||||
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
|
||||
.unwrap_err();
|
||||
assert!(is_unsupported_summary_content_error(&no_artifact));
|
||||
|
||||
add_summary_image_artifact(
|
||||
&paths,
|
||||
entry.id,
|
||||
99,
|
||||
"primary_media",
|
||||
"mp4",
|
||||
"video/mp4",
|
||||
1,
|
||||
);
|
||||
add_summary_image_artifact(&paths, entry.id, 99, "primary_media", "mp4", "video/mp4", 1);
|
||||
let video = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
|
||||
.unwrap_err();
|
||||
assert!(is_unsupported_summary_content_error(&video));
|
||||
|
|
@ -1839,18 +1887,28 @@ mod tests {
|
|||
)
|
||||
.unwrap();
|
||||
drop(conn);
|
||||
let empty_text = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
|
||||
.unwrap_err();
|
||||
let empty_text =
|
||||
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
|
||||
.unwrap_err();
|
||||
assert!(is_unsupported_summary_content_error(&empty_text));
|
||||
|
||||
std::fs::remove_file(paths.store_path.join(empty_relpath)).unwrap();
|
||||
let read_error = build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
|
||||
.unwrap_err();
|
||||
let read_error =
|
||||
build_summary_input(&paths, &entry.entry_uid, SummaryBuildOptions::default())
|
||||
.unwrap_err();
|
||||
assert!(!is_unsupported_summary_content_error(&read_error));
|
||||
assert_eq!(UNSUPPORTED_SUMMARY_CONTENT_MESSAGE, "This entry can’t be summarized yet.\n\nIt doesn’t contain archived text that a summary provider can read. Summaries currently support text notes, web pages, X posts and threads, and X Articles. Video, audio, and image-only entries need a transcript or text source.");
|
||||
assert_eq!(
|
||||
UNSUPPORTED_SUMMARY_CONTENT_MESSAGE,
|
||||
"This entry can’t be summarized yet.\n\nIt doesn’t contain archived text that a summary provider can read. Summaries currently support text notes, web pages, X posts and threads, and X Articles. Video, audio, and image-only entries need a transcript or text source."
|
||||
);
|
||||
|
||||
assert!(!is_unsupported_summary_content_error(&anyhow!("provider timeout")));
|
||||
assert!(!is_unsupported_summary_content_error(&anyhow!("entry not found: {}", entry.entry_uid)));
|
||||
assert!(!is_unsupported_summary_content_error(&anyhow!(
|
||||
"provider timeout"
|
||||
)));
|
||||
assert!(!is_unsupported_summary_content_error(&anyhow!(
|
||||
"entry not found: {}",
|
||||
entry.entry_uid
|
||||
)));
|
||||
}
|
||||
|
||||
fn add_summary_image_artifact(
|
||||
|
|
@ -1917,13 +1975,17 @@ mod tests {
|
|||
let text_only = build_summary_input(
|
||||
&paths,
|
||||
&entry.entry_uid,
|
||||
SummaryBuildOptions { include_images: false },
|
||||
SummaryBuildOptions {
|
||||
include_images: false,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
let visual = build_summary_input(
|
||||
&paths,
|
||||
&entry.entry_uid,
|
||||
SummaryBuildOptions { include_images: true },
|
||||
SummaryBuildOptions {
|
||||
include_images: true,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
|
|
@ -1944,38 +2006,112 @@ mod tests {
|
|||
"summary-image-09",
|
||||
]
|
||||
);
|
||||
assert!(visual
|
||||
.request
|
||||
.images
|
||||
.iter()
|
||||
.all(|image| image.byte_size <= MAX_SUMMARY_IMAGE_BYTES));
|
||||
assert!(
|
||||
visual
|
||||
.request
|
||||
.images
|
||||
.iter()
|
||||
.all(|image| image.byte_size <= MAX_SUMMARY_IMAGE_BYTES)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn summary_image_selection_enforces_aggregate_limit_and_keeps_scanning() {
|
||||
let (_temp, paths, entry) = summary_image_fixture();
|
||||
add_summary_image_artifact(&paths, entry.id, 0, "media", "jpg", "image/jpeg", 4 * 1024 * 1024);
|
||||
add_summary_image_artifact(&paths, entry.id, 1, "media", "png", "image/png", 4 * 1024 * 1024);
|
||||
add_summary_image_artifact(&paths, entry.id, 2, "media", "webp", "image/webp", 4 * 1024 * 1024);
|
||||
add_summary_image_artifact(
|
||||
&paths,
|
||||
entry.id,
|
||||
0,
|
||||
"media",
|
||||
"jpg",
|
||||
"image/jpeg",
|
||||
4 * 1024 * 1024,
|
||||
);
|
||||
add_summary_image_artifact(
|
||||
&paths,
|
||||
entry.id,
|
||||
1,
|
||||
"media",
|
||||
"png",
|
||||
"image/png",
|
||||
4 * 1024 * 1024,
|
||||
);
|
||||
add_summary_image_artifact(
|
||||
&paths,
|
||||
entry.id,
|
||||
2,
|
||||
"media",
|
||||
"webp",
|
||||
"image/webp",
|
||||
4 * 1024 * 1024,
|
||||
);
|
||||
add_summary_image_artifact(&paths, entry.id, 3, "media", "gif", "image/gif", 1);
|
||||
|
||||
let visual = build_summary_input(
|
||||
&paths,
|
||||
&entry.entry_uid,
|
||||
SummaryBuildOptions { include_images: true },
|
||||
SummaryBuildOptions {
|
||||
include_images: true,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(visual.request.images.len(), 3);
|
||||
assert_eq!(
|
||||
visual.request.images.iter().map(|image| image.byte_size).sum::<u64>(),
|
||||
visual
|
||||
.request
|
||||
.images
|
||||
.iter()
|
||||
.map(|image| image.byte_size)
|
||||
.sum::<u64>(),
|
||||
MAX_SUMMARY_IMAGE_TOTAL_BYTES
|
||||
);
|
||||
assert!(visual
|
||||
.request
|
||||
.images
|
||||
.iter()
|
||||
.all(|image| image.sha256 != "summary-image-03"));
|
||||
assert!(
|
||||
visual
|
||||
.request
|
||||
.images
|
||||
.iter()
|
||||
.all(|image| image.sha256 != "summary-image-03")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn summarize_entry_uses_requested_alias_for_cache_and_response_model_for_display() {
|
||||
struct ResolvedModelProvider;
|
||||
|
||||
impl SummaryProvider for ResolvedModelProvider {
|
||||
fn kind(&self) -> &'static str {
|
||||
"anthropic_http"
|
||||
}
|
||||
fn model(&self) -> Option<&str> {
|
||||
Some("claude-3-5-sonnet-latest")
|
||||
}
|
||||
fn summarize(&self, _: &SummaryRequest) -> Result<SummaryOutput> {
|
||||
Ok(SummaryOutput {
|
||||
text: r#"{"tldr":"t","summary":"s","tags":[]}"#.to_string(),
|
||||
model: Some("claude-3-5-sonnet-20241022".to_string()),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
let (_temp, paths, entry) = summary_image_fixture();
|
||||
let record = summarize_entry(
|
||||
&paths,
|
||||
&entry.entry_uid,
|
||||
SummaryBuildOptions::default(),
|
||||
&ResolvedModelProvider,
|
||||
"v1",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
record.provider_model.as_deref(),
|
||||
Some("claude-3-5-sonnet-latest")
|
||||
);
|
||||
assert_eq!(
|
||||
record.resolved_model.as_deref(),
|
||||
Some("claude-3-5-sonnet-20241022")
|
||||
);
|
||||
}
|
||||
|
||||
// ── Output normalization ───────────────────────────────────────────────
|
||||
|
|
@ -1997,7 +2133,8 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn normalize_summary_json_recovers_json_wrapped_in_prose() {
|
||||
let raw = "Sure! Here you go:\n{\"tldr\":\"t\",\"summary\":\"s\",\"tags\":[]}\nHope that helps.";
|
||||
let raw =
|
||||
"Sure! Here you go:\n{\"tldr\":\"t\",\"summary\":\"s\",\"tags\":[]}\nHope that helps.";
|
||||
let v: serde_json::Value = serde_json::from_str(&normalize_summary_json(raw)).unwrap();
|
||||
assert_eq!(v["tldr"], "t");
|
||||
}
|
||||
|
|
@ -2024,13 +2161,17 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn run_cli_kills_a_child_that_overruns_its_timeout() {
|
||||
let err = run_cli(Path::new("sleep"), &["30"], "", 1).unwrap_err().to_string();
|
||||
let err = run_cli(Path::new("sleep"), &["30"], "", 1)
|
||||
.unwrap_err()
|
||||
.to_string();
|
||||
assert!(err.contains("timed out"), "got: {err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_cli_reports_a_nonzero_exit() {
|
||||
let err = run_cli(Path::new("false"), &[], "", 30).unwrap_err().to_string();
|
||||
let err = run_cli(Path::new("false"), &[], "", 30)
|
||||
.unwrap_err()
|
||||
.to_string();
|
||||
assert!(err.contains("exited with"), "got: {err}");
|
||||
}
|
||||
}
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -6,7 +6,7 @@
|
|||
<title>Archivr</title>
|
||||
<link rel="icon" type="image/svg+xml" href="/favicon.svg">
|
||||
<link rel="icon" type="image/x-icon" href="/favicon.ico">
|
||||
<script type="module" crossorigin src="/assets/index-DH0j8cUl.js"></script>
|
||||
<script type="module" crossorigin src="/assets/index-DMKGhxnA.js"></script>
|
||||
<link rel="stylesheet" crossorigin href="/assets/index-1h0SqIvL.css">
|
||||
</head>
|
||||
<body>
|
||||
|
|
|
|||
|
|
@ -590,7 +590,9 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
)}
|
||||
<p className="rail-summary-provider">
|
||||
{PROVIDER_LABEL[summary.provider_kind] || summary.provider_kind}
|
||||
{summary.provider_model ? ` \u00b7 ${summary.provider_model}` : ''}
|
||||
{summary.resolved_model || summary.provider_model
|
||||
? ` \u00b7 ${summary.resolved_model || summary.provider_model}`
|
||||
: ''}
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue