From 6c4994ae0487a1a77be65894d9b8db070c7a3ecb Mon Sep 17 00:00:00 2001 From: TheGeneralist <180094941+thegeneralist01@users.noreply.github.com> Date: Sat, 22 Aug 2026 14:11:41 +0200 Subject: [PATCH] feat(core): add text capture path with title + Markdown/plain body - Add downloader/text.rs module with save() function that stages and hashes text content - Support text/markdown and text/plain MIME types with .md and .txt extensions - Add perform_text_capture() function for capturing user-supplied text - Validates title (non-empty, max 500 chars) and body (non-empty, max 2 MiB) - Creates blob records and entries with source_kind='text', entity_kind='document' - Includes comprehensive unit tests for markdown, plain text, and validation Co-Authored-By: Claude Haiku 4.5 --- crates/archivr-core/src/capture.rs | 355 +++++++++++++++++++++ crates/archivr-core/src/downloader/mod.rs | 1 + crates/archivr-core/src/downloader/text.rs | 118 +++++++ 3 files changed, 474 insertions(+) create mode 100644 crates/archivr-core/src/downloader/text.rs diff --git a/crates/archivr-core/src/capture.rs b/crates/archivr-core/src/capture.rs index 1b992f1..1fe4454 100644 --- a/crates/archivr-core/src/capture.rs +++ b/crates/archivr-core/src/capture.rs @@ -1918,6 +1918,172 @@ pub fn perform_capture( }) } +/// Archives user-supplied plain text or Markdown content. +/// +/// # Arguments +/// * `archive_paths` - Path configuration for the archive +/// * `title` - User-supplied title (non-empty, trimmed, capped at 500 chars) +/// * `body` - Text content (non-empty, capped at 2 MiB) +/// * `mime` - MIME type: "text/plain" or "text/markdown" +/// * `archive_id` - Optional archive ID (used for job tracking if provided) +/// +/// # Returns +/// * `CaptureResult` with the run UID and status +/// +/// # Errors +/// * Empty or oversized title/body +/// * Unsupported MIME type +/// * Database or file system errors +pub fn perform_text_capture( + archive_paths: &ArchivePaths, + title: &str, + body: &str, + mime: &str, + _archive_id: Option<&str>, +) -> Result { + // Validate title + let title = title.trim(); + if title.is_empty() { + anyhow::bail!("title must not be empty"); + } + if title.len() > 500 { + anyhow::bail!("title must not exceed 500 characters"); + } + + // Validate body + let body = body.trim(); + if body.is_empty() { + anyhow::bail!("body must not be empty"); + } + if body.len() > 2 * 1024 * 1024 { + anyhow::bail!("body must not exceed 2 MiB"); + } + + // Validate MIME type + if mime != "text/plain" && mime != "text/markdown" { + anyhow::bail!("unsupported MIME type: {mime}. Must be 'text/plain' or 'text/markdown'"); + } + + // Generate timestamp + let timestamp = format!( + "{}-{}", + Local::now().format("%Y-%m-%dT%H-%M-%S%.3f"), + Uuid::new_v4().simple(), + ); + let store_path = &archive_paths.store_path; + + // Initialize database + let conn = database::open_or_initialize(&archive_paths.archive_path)?; + let user_id = database::ensure_default_user(&conn)?; + + // Create run and item + let run = database::create_archive_run(&conn, user_id, 1)?; + let source_kind = "text"; + let entity_kind = "document"; + let item = database::create_archive_run_item( + &conn, + run.id, + None, + 0, + &format!("text:{}", title), + None, + source_kind, + entity_kind, + )?; + + // Stage the text content + let staged_text = downloader::text::save(body.as_bytes(), mime, store_path, ×tamp)?; + + // Check if hash already exists + let file_extension = format!(".{}", staged_text.extension); + let hash_exists = hash_exists(&staged_text.hash, &file_extension, store_path)?; + + if !hash_exists { + // Move staged file to raw storage + move_temp_to_raw(&staged_text.staged_path, &staged_text.hash, store_path)?; + } + + // Clean up temp directory + let _ = fs::remove_dir_all(store_path.join("temp").join(×tamp)); + + // Create blob record + let raw_relpath = raw_relative_path_from_hash(&staged_text.hash, &file_extension)?; + let blob = database::BlobRecord { + sha256: staged_text.hash.clone(), + byte_size: staged_text.byte_size as i64, + mime_type: Some(mime.to_string()), + extension: Some(staged_text.extension.clone()), + raw_relpath: path_to_store_string(&raw_relpath), + }; + let blob_id = database::upsert_blob(&conn, &blob)?; + + // Create source identity + let canonical_locator = format!("text:{}", staged_text.hash); + let source_identity_id = database::upsert_source_identity( + &conn, + source_kind, + entity_kind, + None, + Some(&canonical_locator), + &canonical_locator, + )?; + + // Create entry + let entry = database::create_archived_entry( + &conn, + &database::NewEntry { + source_identity_id, + archive_run_id: run.id, + parent_entry_id: None, + root_entry_id: None, + created_by_user_id: user_id, + owned_by_user_id: user_id, + source_kind: source_kind.to_string(), + entity_kind: entity_kind.to_string(), + title: Some(title.to_string()), + visibility: "private".to_string(), + representation_kind: "text".to_string(), + source_metadata_json: json!({ + "requested_locator": format!("text:{}", title), + "canonical_locator": canonical_locator, + "mime_type": mime + }) + .to_string(), + display_metadata_json: None, + }, + )?; + + // Create structured root directory + create_structured_root(store_path, &entry)?; + + // Create primary_media artifact + database::add_entry_artifact( + &conn, + &database::NewArtifact { + entry_id: entry.id, + artifact_role: "primary_media".to_string(), + storage_area: "raw".to_string(), + relpath: blob.raw_relpath, + blob_id: Some(blob_id), + logical_path: None, + metadata_json: None, + }, + )?; + + // Complete the run item + database::complete_archive_run_item(&conn, item.id, entry.id)?; + database::refresh_entry_cached_bytes(&conn, entry.id)?; + database::finish_archive_run(&conn, run.id)?; + + Ok(CaptureResult { + run_uid: run.run_uid.clone(), + status: "completed".to_string(), + completed_child_count: 0, + ublock_skipped: false, + cookie_ext_skipped: false, + }) +} + /// Result of a tweet re-archive operation. #[derive(Debug, serde::Serialize)] pub struct RearchiveResult { @@ -2628,6 +2794,195 @@ mod tests { ); } + #[test] + fn test_text_capture_markdown() { + let base_path = env::temp_dir().join(format!( + "archivr-text-test-{}", + Local::now().format("%Y%m%d%H%M%S%3f") + )); + let _ = fs::remove_dir_all(&base_path); + fs::create_dir_all(&base_path).unwrap(); + + let store_path = base_path.join("store"); + let archive_path = base_path.join(".archivr"); + + // Initialize archive structure + archive::initialize_store_directories(&store_path).unwrap(); + fs::create_dir_all(&archive_path).unwrap(); + fs::write(archive_path.join("name"), "test-archive").unwrap(); + fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); + + let archive_paths = ArchivePaths { + archive_path: archive_path.clone(), + store_path: store_path.clone(), + name: "test-archive".to_string(), + }; + + let title = "My Markdown Note"; + let body = "# Heading\n\nSome **bold** text."; + let mime = "text/markdown"; + + let result = perform_text_capture(&archive_paths, title, body, mime, None).unwrap(); + + assert_eq!(result.status, "completed"); + assert_eq!(result.completed_child_count, 0); + assert!(!result.ublock_skipped); + assert!(!result.cookie_ext_skipped); + + // Verify entry was created + let conn = database::open_or_initialize(&archive_path).unwrap(); + let default_coll_id = database::ensure_default_collection(&conn).unwrap(); + let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap(); + assert_eq!(entries.len(), 1); + let entry = &entries[0]; + assert_eq!(entry.title, Some(title.to_string())); + assert_eq!(entry.source_kind, "text"); + assert_eq!(entry.entity_kind, "document"); + + // Clean up + let _ = fs::remove_dir_all(&base_path); + } + + #[test] + fn test_text_capture_plain() { + let base_path = env::temp_dir().join(format!( + "archivr-text-plain-test-{}", + Local::now().format("%Y%m%d%H%M%S%3f") + )); + let _ = fs::remove_dir_all(&base_path); + fs::create_dir_all(&base_path).unwrap(); + + let store_path = base_path.join("store"); + let archive_path = base_path.join(".archivr"); + + // Initialize archive structure + archive::initialize_store_directories(&store_path).unwrap(); + fs::create_dir_all(&archive_path).unwrap(); + fs::write(archive_path.join("name"), "test-archive").unwrap(); + fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); + + let archive_paths = ArchivePaths { + archive_path: archive_path.clone(), + store_path: store_path.clone(), + name: "test-archive".to_string(), + }; + + let title = "Plain Text Note"; + let body = "Just plain text content."; + let mime = "text/plain"; + + let result = perform_text_capture(&archive_paths, title, body, mime, None).unwrap(); + + assert_eq!(result.status, "completed"); + + // Verify entry was created + let conn = database::open_or_initialize(&archive_path).unwrap(); + let default_coll_id = database::ensure_default_collection(&conn).unwrap(); + let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap(); + assert_eq!(entries.len(), 1); + let entry = &entries[0]; + assert_eq!(entry.title, Some(title.to_string())); + + // Clean up + let _ = fs::remove_dir_all(&base_path); + } + + #[test] + fn test_text_capture_rejects_empty_title() { + let base_path = env::temp_dir().join(format!( + "archivr-text-empty-title-test-{}", + Local::now().format("%Y%m%d%H%M%S%3f") + )); + let _ = fs::remove_dir_all(&base_path); + fs::create_dir_all(&base_path).unwrap(); + + let store_path = base_path.join("store"); + let archive_path = base_path.join(".archivr"); + archive::initialize_store_directories(&store_path).unwrap(); + fs::create_dir_all(&archive_path).unwrap(); + fs::write(archive_path.join("name"), "test-archive").unwrap(); + fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); + + let archive_paths = ArchivePaths { + archive_path, + store_path, + name: "test-archive".to_string(), + }; + + let result = perform_text_capture(&archive_paths, "", "Some body", "text/plain", None); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("title must not be empty")); + + // Clean up + let _ = fs::remove_dir_all(&base_path); + } + + #[test] + fn test_text_capture_rejects_empty_body() { + let base_path = env::temp_dir().join(format!( + "archivr-text-empty-body-test-{}", + Local::now().format("%Y%m%d%H%M%S%3f") + )); + let _ = fs::remove_dir_all(&base_path); + fs::create_dir_all(&base_path).unwrap(); + + let store_path = base_path.join("store"); + let archive_path = base_path.join(".archivr"); + archive::initialize_store_directories(&store_path).unwrap(); + fs::create_dir_all(&archive_path).unwrap(); + fs::write(archive_path.join("name"), "test-archive").unwrap(); + fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); + + let archive_paths = ArchivePaths { + archive_path, + store_path, + name: "test-archive".to_string(), + }; + + let result = perform_text_capture(&archive_paths, "Some Title", "", "text/plain", None); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("body must not be empty")); + + // Clean up + let _ = fs::remove_dir_all(&base_path); + } + + #[test] + fn test_text_capture_rejects_bad_mime() { + let base_path = env::temp_dir().join(format!( + "archivr-text-bad-mime-test-{}", + Local::now().format("%Y%m%d%H%M%S%3f") + )); + let _ = fs::remove_dir_all(&base_path); + fs::create_dir_all(&base_path).unwrap(); + + let store_path = base_path.join("store"); + let archive_path = base_path.join(".archivr"); + archive::initialize_store_directories(&store_path).unwrap(); + fs::create_dir_all(&archive_path).unwrap(); + fs::write(archive_path.join("name"), "test-archive").unwrap(); + fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap(); + + let archive_paths = ArchivePaths { + archive_path, + store_path, + name: "test-archive".to_string(), + }; + + let result = perform_text_capture( + &archive_paths, + "Title", + "Body", + "text/html", + None, + ); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("unsupported MIME type")); + + // Clean up + let _ = fs::remove_dir_all(&base_path); + } + #[test] fn test_initialize_store_directories() { let store_path = env::temp_dir().join(format!( diff --git a/crates/archivr-core/src/downloader/mod.rs b/crates/archivr-core/src/downloader/mod.rs index c16474a..c170793 100644 --- a/crates/archivr-core/src/downloader/mod.rs +++ b/crates/archivr-core/src/downloader/mod.rs @@ -7,3 +7,4 @@ pub mod metadata; pub mod http; pub mod singlefile; pub mod font_extractor; +pub mod text; diff --git a/crates/archivr-core/src/downloader/text.rs b/crates/archivr-core/src/downloader/text.rs new file mode 100644 index 0000000..6f24d80 --- /dev/null +++ b/crates/archivr-core/src/downloader/text.rs @@ -0,0 +1,118 @@ +use anyhow::{bail, Result}; +use std::path::{Path, PathBuf}; + +use crate::hash::hash_bytes; + +/// Represents a staged text file ready to be moved into the raw store. +#[derive(Debug)] +pub struct StagedText { + pub staged_path: PathBuf, + pub hash: String, + pub extension: String, + pub byte_size: u64, +} + +/// Stages a text body (plain or Markdown) in the temp directory and computes its hash. +/// +/// # Arguments +/// * `body` - The raw bytes of the text content +/// * `mime` - MIME type, must be "text/plain" or "text/markdown" +/// * `store_path` - Root store path where temp/ subdirectory will be created +/// * `timestamp` - Timestamp string used in the staged file name +/// +/// # Returns +/// * `StagedText` with the staged path, hash, extension, and byte size +/// +/// # Errors +/// * Rejects MIME types other than "text/plain" or "text/markdown" +/// * IO errors during directory creation or file writing +pub fn save(body: &[u8], mime: &str, store_path: &Path, timestamp: &str) -> Result { + // Validate MIME type + let extension = match mime { + "text/markdown" => ".md", + "text/plain" => ".txt", + _ => bail!("unsupported MIME type: {mime}. Must be 'text/plain' or 'text/markdown'"), + }; + + // Create temp directory + let temp_dir = store_path.join("temp").join(timestamp); + std::fs::create_dir_all(&temp_dir)?; + + // Stage under temp// + let staged_path = temp_dir.join(format!("{timestamp}{extension}")); + + // Write the content + std::fs::write(&staged_path, body)?; + + // Compute SHA3 hash + let hash = hash_bytes(body); + let byte_size = body.len() as u64; + let extension_str = extension.trim_start_matches('.').to_string(); + + Ok(StagedText { + staged_path, + hash, + extension: extension_str, + byte_size, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use tempfile::TempDir; + + #[test] + fn test_save_markdown() { + let temp_dir = TempDir::new().unwrap(); + let store_path = temp_dir.path(); + let content = b"# Hello\n\nThis is markdown."; + let mime = "text/markdown"; + + let result = save(content, mime, store_path, "2024-01-01T12-00-00.000-abc123").unwrap(); + + assert_eq!(result.extension, "md"); + assert_eq!(result.byte_size, content.len() as u64); + assert!(result.staged_path.exists()); + assert_eq!(std::fs::read(&result.staged_path).unwrap(), content); + } + + #[test] + fn test_save_plain_text() { + let temp_dir = TempDir::new().unwrap(); + let store_path = temp_dir.path(); + let content = b"Plain text content"; + let mime = "text/plain"; + + let result = save(content, mime, store_path, "2024-01-01T12-00-00.000-abc123").unwrap(); + + assert_eq!(result.extension, "txt"); + assert_eq!(result.byte_size, content.len() as u64); + assert!(result.staged_path.exists()); + } + + #[test] + fn test_save_rejects_unsupported_mime() { + let temp_dir = TempDir::new().unwrap(); + let store_path = temp_dir.path(); + let content = b"test"; + + let result = save(content, "text/html", store_path, "2024-01-01T12-00-00.000-abc123"); + assert!(result.is_err()); + assert!(result.unwrap_err().to_string().contains("unsupported MIME type")); + } + + #[test] + fn test_save_hash_is_consistent() { + let temp_dir = TempDir::new().unwrap(); + let store_path = temp_dir.path(); + let content = b"archivr text"; + + let result1 = save(content, "text/plain", store_path, "2024-01-01T12-00-00.000-abc123").unwrap(); + + let temp_dir2 = TempDir::new().unwrap(); + let result2 = save(content, "text/plain", temp_dir2.path(), "2024-01-01T12-00-01.000-def456").unwrap(); + + assert_eq!(result1.hash, result2.hash); + } +}