mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 12:55:00 +02:00
feat(core): add text capture path with title + Markdown/plain body
- Add downloader/text.rs module with save() function that stages and hashes text content - Support text/markdown and text/plain MIME types with .md and .txt extensions - Add perform_text_capture() function for capturing user-supplied text - Validates title (non-empty, max 500 chars) and body (non-empty, max 2 MiB) - Creates blob records and entries with source_kind='text', entity_kind='document' - Includes comprehensive unit tests for markdown, plain text, and validation Co-Authored-By: Claude Haiku 4.5 <noreply@anthropic.com>
This commit is contained in:
parent
bf5f95397f
commit
6c4994ae04
3 changed files with 474 additions and 0 deletions
|
|
@ -1918,6 +1918,172 @@ pub fn perform_capture(
|
|||
})
|
||||
}
|
||||
|
||||
/// Archives user-supplied plain text or Markdown content.
|
||||
///
|
||||
/// # Arguments
|
||||
/// * `archive_paths` - Path configuration for the archive
|
||||
/// * `title` - User-supplied title (non-empty, trimmed, capped at 500 chars)
|
||||
/// * `body` - Text content (non-empty, capped at 2 MiB)
|
||||
/// * `mime` - MIME type: "text/plain" or "text/markdown"
|
||||
/// * `archive_id` - Optional archive ID (used for job tracking if provided)
|
||||
///
|
||||
/// # Returns
|
||||
/// * `CaptureResult` with the run UID and status
|
||||
///
|
||||
/// # Errors
|
||||
/// * Empty or oversized title/body
|
||||
/// * Unsupported MIME type
|
||||
/// * Database or file system errors
|
||||
pub fn perform_text_capture(
|
||||
archive_paths: &ArchivePaths,
|
||||
title: &str,
|
||||
body: &str,
|
||||
mime: &str,
|
||||
_archive_id: Option<&str>,
|
||||
) -> Result<CaptureResult> {
|
||||
// Validate title
|
||||
let title = title.trim();
|
||||
if title.is_empty() {
|
||||
anyhow::bail!("title must not be empty");
|
||||
}
|
||||
if title.len() > 500 {
|
||||
anyhow::bail!("title must not exceed 500 characters");
|
||||
}
|
||||
|
||||
// Validate body
|
||||
let body = body.trim();
|
||||
if body.is_empty() {
|
||||
anyhow::bail!("body must not be empty");
|
||||
}
|
||||
if body.len() > 2 * 1024 * 1024 {
|
||||
anyhow::bail!("body must not exceed 2 MiB");
|
||||
}
|
||||
|
||||
// Validate MIME type
|
||||
if mime != "text/plain" && mime != "text/markdown" {
|
||||
anyhow::bail!("unsupported MIME type: {mime}. Must be 'text/plain' or 'text/markdown'");
|
||||
}
|
||||
|
||||
// Generate timestamp
|
||||
let timestamp = format!(
|
||||
"{}-{}",
|
||||
Local::now().format("%Y-%m-%dT%H-%M-%S%.3f"),
|
||||
Uuid::new_v4().simple(),
|
||||
);
|
||||
let store_path = &archive_paths.store_path;
|
||||
|
||||
// Initialize database
|
||||
let conn = database::open_or_initialize(&archive_paths.archive_path)?;
|
||||
let user_id = database::ensure_default_user(&conn)?;
|
||||
|
||||
// Create run and item
|
||||
let run = database::create_archive_run(&conn, user_id, 1)?;
|
||||
let source_kind = "text";
|
||||
let entity_kind = "document";
|
||||
let item = database::create_archive_run_item(
|
||||
&conn,
|
||||
run.id,
|
||||
None,
|
||||
0,
|
||||
&format!("text:{}", title),
|
||||
None,
|
||||
source_kind,
|
||||
entity_kind,
|
||||
)?;
|
||||
|
||||
// Stage the text content
|
||||
let staged_text = downloader::text::save(body.as_bytes(), mime, store_path, ×tamp)?;
|
||||
|
||||
// Check if hash already exists
|
||||
let file_extension = format!(".{}", staged_text.extension);
|
||||
let hash_exists = hash_exists(&staged_text.hash, &file_extension, store_path)?;
|
||||
|
||||
if !hash_exists {
|
||||
// Move staged file to raw storage
|
||||
move_temp_to_raw(&staged_text.staged_path, &staged_text.hash, store_path)?;
|
||||
}
|
||||
|
||||
// Clean up temp directory
|
||||
let _ = fs::remove_dir_all(store_path.join("temp").join(×tamp));
|
||||
|
||||
// Create blob record
|
||||
let raw_relpath = raw_relative_path_from_hash(&staged_text.hash, &file_extension)?;
|
||||
let blob = database::BlobRecord {
|
||||
sha256: staged_text.hash.clone(),
|
||||
byte_size: staged_text.byte_size as i64,
|
||||
mime_type: Some(mime.to_string()),
|
||||
extension: Some(staged_text.extension.clone()),
|
||||
raw_relpath: path_to_store_string(&raw_relpath),
|
||||
};
|
||||
let blob_id = database::upsert_blob(&conn, &blob)?;
|
||||
|
||||
// Create source identity
|
||||
let canonical_locator = format!("text:{}", staged_text.hash);
|
||||
let source_identity_id = database::upsert_source_identity(
|
||||
&conn,
|
||||
source_kind,
|
||||
entity_kind,
|
||||
None,
|
||||
Some(&canonical_locator),
|
||||
&canonical_locator,
|
||||
)?;
|
||||
|
||||
// Create entry
|
||||
let entry = database::create_archived_entry(
|
||||
&conn,
|
||||
&database::NewEntry {
|
||||
source_identity_id,
|
||||
archive_run_id: run.id,
|
||||
parent_entry_id: None,
|
||||
root_entry_id: None,
|
||||
created_by_user_id: user_id,
|
||||
owned_by_user_id: user_id,
|
||||
source_kind: source_kind.to_string(),
|
||||
entity_kind: entity_kind.to_string(),
|
||||
title: Some(title.to_string()),
|
||||
visibility: "private".to_string(),
|
||||
representation_kind: "text".to_string(),
|
||||
source_metadata_json: json!({
|
||||
"requested_locator": format!("text:{}", title),
|
||||
"canonical_locator": canonical_locator,
|
||||
"mime_type": mime
|
||||
})
|
||||
.to_string(),
|
||||
display_metadata_json: None,
|
||||
},
|
||||
)?;
|
||||
|
||||
// Create structured root directory
|
||||
create_structured_root(store_path, &entry)?;
|
||||
|
||||
// Create primary_media artifact
|
||||
database::add_entry_artifact(
|
||||
&conn,
|
||||
&database::NewArtifact {
|
||||
entry_id: entry.id,
|
||||
artifact_role: "primary_media".to_string(),
|
||||
storage_area: "raw".to_string(),
|
||||
relpath: blob.raw_relpath,
|
||||
blob_id: Some(blob_id),
|
||||
logical_path: None,
|
||||
metadata_json: None,
|
||||
},
|
||||
)?;
|
||||
|
||||
// Complete the run item
|
||||
database::complete_archive_run_item(&conn, item.id, entry.id)?;
|
||||
database::refresh_entry_cached_bytes(&conn, entry.id)?;
|
||||
database::finish_archive_run(&conn, run.id)?;
|
||||
|
||||
Ok(CaptureResult {
|
||||
run_uid: run.run_uid.clone(),
|
||||
status: "completed".to_string(),
|
||||
completed_child_count: 0,
|
||||
ublock_skipped: false,
|
||||
cookie_ext_skipped: false,
|
||||
})
|
||||
}
|
||||
|
||||
/// Result of a tweet re-archive operation.
|
||||
#[derive(Debug, serde::Serialize)]
|
||||
pub struct RearchiveResult {
|
||||
|
|
@ -2628,6 +2794,195 @@ mod tests {
|
|||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_text_capture_markdown() {
|
||||
let base_path = env::temp_dir().join(format!(
|
||||
"archivr-text-test-{}",
|
||||
Local::now().format("%Y%m%d%H%M%S%3f")
|
||||
));
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
fs::create_dir_all(&base_path).unwrap();
|
||||
|
||||
let store_path = base_path.join("store");
|
||||
let archive_path = base_path.join(".archivr");
|
||||
|
||||
// Initialize archive structure
|
||||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path: archive_path.clone(),
|
||||
store_path: store_path.clone(),
|
||||
name: "test-archive".to_string(),
|
||||
};
|
||||
|
||||
let title = "My Markdown Note";
|
||||
let body = "# Heading\n\nSome **bold** text.";
|
||||
let mime = "text/markdown";
|
||||
|
||||
let result = perform_text_capture(&archive_paths, title, body, mime, None).unwrap();
|
||||
|
||||
assert_eq!(result.status, "completed");
|
||||
assert_eq!(result.completed_child_count, 0);
|
||||
assert!(!result.ublock_skipped);
|
||||
assert!(!result.cookie_ext_skipped);
|
||||
|
||||
// Verify entry was created
|
||||
let conn = database::open_or_initialize(&archive_path).unwrap();
|
||||
let default_coll_id = database::ensure_default_collection(&conn).unwrap();
|
||||
let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
|
||||
assert_eq!(entries.len(), 1);
|
||||
let entry = &entries[0];
|
||||
assert_eq!(entry.title, Some(title.to_string()));
|
||||
assert_eq!(entry.source_kind, "text");
|
||||
assert_eq!(entry.entity_kind, "document");
|
||||
|
||||
// Clean up
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_text_capture_plain() {
|
||||
let base_path = env::temp_dir().join(format!(
|
||||
"archivr-text-plain-test-{}",
|
||||
Local::now().format("%Y%m%d%H%M%S%3f")
|
||||
));
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
fs::create_dir_all(&base_path).unwrap();
|
||||
|
||||
let store_path = base_path.join("store");
|
||||
let archive_path = base_path.join(".archivr");
|
||||
|
||||
// Initialize archive structure
|
||||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path: archive_path.clone(),
|
||||
store_path: store_path.clone(),
|
||||
name: "test-archive".to_string(),
|
||||
};
|
||||
|
||||
let title = "Plain Text Note";
|
||||
let body = "Just plain text content.";
|
||||
let mime = "text/plain";
|
||||
|
||||
let result = perform_text_capture(&archive_paths, title, body, mime, None).unwrap();
|
||||
|
||||
assert_eq!(result.status, "completed");
|
||||
|
||||
// Verify entry was created
|
||||
let conn = database::open_or_initialize(&archive_path).unwrap();
|
||||
let default_coll_id = database::ensure_default_collection(&conn).unwrap();
|
||||
let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
|
||||
assert_eq!(entries.len(), 1);
|
||||
let entry = &entries[0];
|
||||
assert_eq!(entry.title, Some(title.to_string()));
|
||||
|
||||
// Clean up
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_text_capture_rejects_empty_title() {
|
||||
let base_path = env::temp_dir().join(format!(
|
||||
"archivr-text-empty-title-test-{}",
|
||||
Local::now().format("%Y%m%d%H%M%S%3f")
|
||||
));
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
fs::create_dir_all(&base_path).unwrap();
|
||||
|
||||
let store_path = base_path.join("store");
|
||||
let archive_path = base_path.join(".archivr");
|
||||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path,
|
||||
store_path,
|
||||
name: "test-archive".to_string(),
|
||||
};
|
||||
|
||||
let result = perform_text_capture(&archive_paths, "", "Some body", "text/plain", None);
|
||||
assert!(result.is_err());
|
||||
assert!(result.unwrap_err().to_string().contains("title must not be empty"));
|
||||
|
||||
// Clean up
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_text_capture_rejects_empty_body() {
|
||||
let base_path = env::temp_dir().join(format!(
|
||||
"archivr-text-empty-body-test-{}",
|
||||
Local::now().format("%Y%m%d%H%M%S%3f")
|
||||
));
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
fs::create_dir_all(&base_path).unwrap();
|
||||
|
||||
let store_path = base_path.join("store");
|
||||
let archive_path = base_path.join(".archivr");
|
||||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path,
|
||||
store_path,
|
||||
name: "test-archive".to_string(),
|
||||
};
|
||||
|
||||
let result = perform_text_capture(&archive_paths, "Some Title", "", "text/plain", None);
|
||||
assert!(result.is_err());
|
||||
assert!(result.unwrap_err().to_string().contains("body must not be empty"));
|
||||
|
||||
// Clean up
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_text_capture_rejects_bad_mime() {
|
||||
let base_path = env::temp_dir().join(format!(
|
||||
"archivr-text-bad-mime-test-{}",
|
||||
Local::now().format("%Y%m%d%H%M%S%3f")
|
||||
));
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
fs::create_dir_all(&base_path).unwrap();
|
||||
|
||||
let store_path = base_path.join("store");
|
||||
let archive_path = base_path.join(".archivr");
|
||||
archive::initialize_store_directories(&store_path).unwrap();
|
||||
fs::create_dir_all(&archive_path).unwrap();
|
||||
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||
|
||||
let archive_paths = ArchivePaths {
|
||||
archive_path,
|
||||
store_path,
|
||||
name: "test-archive".to_string(),
|
||||
};
|
||||
|
||||
let result = perform_text_capture(
|
||||
&archive_paths,
|
||||
"Title",
|
||||
"Body",
|
||||
"text/html",
|
||||
None,
|
||||
);
|
||||
assert!(result.is_err());
|
||||
assert!(result.unwrap_err().to_string().contains("unsupported MIME type"));
|
||||
|
||||
// Clean up
|
||||
let _ = fs::remove_dir_all(&base_path);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_initialize_store_directories() {
|
||||
let store_path = env::temp_dir().join(format!(
|
||||
|
|
|
|||
|
|
@ -7,3 +7,4 @@ pub mod metadata;
|
|||
pub mod http;
|
||||
pub mod singlefile;
|
||||
pub mod font_extractor;
|
||||
pub mod text;
|
||||
|
|
|
|||
118
crates/archivr-core/src/downloader/text.rs
Normal file
118
crates/archivr-core/src/downloader/text.rs
Normal file
|
|
@ -0,0 +1,118 @@
|
|||
use anyhow::{bail, Result};
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::hash::hash_bytes;
|
||||
|
||||
/// Represents a staged text file ready to be moved into the raw store.
|
||||
#[derive(Debug)]
|
||||
pub struct StagedText {
|
||||
pub staged_path: PathBuf,
|
||||
pub hash: String,
|
||||
pub extension: String,
|
||||
pub byte_size: u64,
|
||||
}
|
||||
|
||||
/// Stages a text body (plain or Markdown) in the temp directory and computes its hash.
|
||||
///
|
||||
/// # Arguments
|
||||
/// * `body` - The raw bytes of the text content
|
||||
/// * `mime` - MIME type, must be "text/plain" or "text/markdown"
|
||||
/// * `store_path` - Root store path where temp/ subdirectory will be created
|
||||
/// * `timestamp` - Timestamp string used in the staged file name
|
||||
///
|
||||
/// # Returns
|
||||
/// * `StagedText` with the staged path, hash, extension, and byte size
|
||||
///
|
||||
/// # Errors
|
||||
/// * Rejects MIME types other than "text/plain" or "text/markdown"
|
||||
/// * IO errors during directory creation or file writing
|
||||
pub fn save(body: &[u8], mime: &str, store_path: &Path, timestamp: &str) -> Result<StagedText> {
|
||||
// Validate MIME type
|
||||
let extension = match mime {
|
||||
"text/markdown" => ".md",
|
||||
"text/plain" => ".txt",
|
||||
_ => bail!("unsupported MIME type: {mime}. Must be 'text/plain' or 'text/markdown'"),
|
||||
};
|
||||
|
||||
// Create temp directory
|
||||
let temp_dir = store_path.join("temp").join(timestamp);
|
||||
std::fs::create_dir_all(&temp_dir)?;
|
||||
|
||||
// Stage under temp/<timestamp>/<timestamp><ext>
|
||||
let staged_path = temp_dir.join(format!("{timestamp}{extension}"));
|
||||
|
||||
// Write the content
|
||||
std::fs::write(&staged_path, body)?;
|
||||
|
||||
// Compute SHA3 hash
|
||||
let hash = hash_bytes(body);
|
||||
let byte_size = body.len() as u64;
|
||||
let extension_str = extension.trim_start_matches('.').to_string();
|
||||
|
||||
Ok(StagedText {
|
||||
staged_path,
|
||||
hash,
|
||||
extension: extension_str,
|
||||
byte_size,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
fn test_save_markdown() {
|
||||
let temp_dir = TempDir::new().unwrap();
|
||||
let store_path = temp_dir.path();
|
||||
let content = b"# Hello\n\nThis is markdown.";
|
||||
let mime = "text/markdown";
|
||||
|
||||
let result = save(content, mime, store_path, "2024-01-01T12-00-00.000-abc123").unwrap();
|
||||
|
||||
assert_eq!(result.extension, "md");
|
||||
assert_eq!(result.byte_size, content.len() as u64);
|
||||
assert!(result.staged_path.exists());
|
||||
assert_eq!(std::fs::read(&result.staged_path).unwrap(), content);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_save_plain_text() {
|
||||
let temp_dir = TempDir::new().unwrap();
|
||||
let store_path = temp_dir.path();
|
||||
let content = b"Plain text content";
|
||||
let mime = "text/plain";
|
||||
|
||||
let result = save(content, mime, store_path, "2024-01-01T12-00-00.000-abc123").unwrap();
|
||||
|
||||
assert_eq!(result.extension, "txt");
|
||||
assert_eq!(result.byte_size, content.len() as u64);
|
||||
assert!(result.staged_path.exists());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_save_rejects_unsupported_mime() {
|
||||
let temp_dir = TempDir::new().unwrap();
|
||||
let store_path = temp_dir.path();
|
||||
let content = b"test";
|
||||
|
||||
let result = save(content, "text/html", store_path, "2024-01-01T12-00-00.000-abc123");
|
||||
assert!(result.is_err());
|
||||
assert!(result.unwrap_err().to_string().contains("unsupported MIME type"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_save_hash_is_consistent() {
|
||||
let temp_dir = TempDir::new().unwrap();
|
||||
let store_path = temp_dir.path();
|
||||
let content = b"archivr text";
|
||||
|
||||
let result1 = save(content, "text/plain", store_path, "2024-01-01T12-00-00.000-abc123").unwrap();
|
||||
|
||||
let temp_dir2 = TempDir::new().unwrap();
|
||||
let result2 = save(content, "text/plain", temp_dir2.path(), "2024-01-01T12-00-01.000-def456").unwrap();
|
||||
|
||||
assert_eq!(result1.hash, result2.hash);
|
||||
}
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue