1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-10-09 12:55:00 +02:00

fix: preserve text capture bytes and hide synthetic URL

This commit is contained in:
archivr-qa 2026-08-24 15:30:56 +02:00
parent 492b1a3168
commit 771c64a811
No known key found for this signature in database
2 changed files with 173 additions and 4 deletions

View file

@ -1951,8 +1951,7 @@ pub fn perform_text_capture(
} }
// Validate body // Validate body
let body = body.trim(); if body.trim().is_empty() {
if body.is_empty() {
anyhow::bail!("body must not be empty"); anyhow::bail!("body must not be empty");
} }
if body.len() > 2 * 1024 * 1024 { if body.len() > 2 * 1024 * 1024 {
@ -2024,7 +2023,7 @@ pub fn perform_text_capture(
source_kind, source_kind,
entity_kind, entity_kind,
None, None,
Some(&canonical_locator), None,
&canonical_locator, &canonical_locator,
)?; )?;
@ -2887,6 +2886,111 @@ mod tests {
let _ = fs::remove_dir_all(&base_path); let _ = fs::remove_dir_all(&base_path);
} }
#[test]
fn test_text_capture_preserves_intentional_whitespace_in_stored_artifact() {
let base_path = env::temp_dir().join(format!(
"archivr-text-whitespace-test-{}",
Local::now().format("%Y%m%d%H%M%S%3f")
));
let _ = fs::remove_dir_all(&base_path);
fs::create_dir_all(&base_path).unwrap();
let store_path = base_path.join("store");
let archive_path = base_path.join(".archivr");
archive::initialize_store_directories(&store_path).unwrap();
fs::create_dir_all(&archive_path).unwrap();
fs::write(archive_path.join("name"), "test-archive").unwrap();
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
let archive_paths = ArchivePaths {
archive_path: archive_path.clone(),
store_path: store_path.clone(),
name: "test-archive".to_string(),
};
let body = " \n# Heading\n\nContent with a final newline\n\t ";
perform_text_capture(
&archive_paths,
"Whitespace Note",
body,
"text/markdown",
None,
)
.unwrap();
let conn = database::open_or_initialize(&archive_path).unwrap();
let raw_relpath: String = conn
.query_row(
"SELECT b.raw_relpath
FROM entry_artifacts ea
JOIN blobs b ON b.id = ea.blob_id
WHERE ea.artifact_role = 'primary_media'",
[],
|row| row.get(0),
)
.unwrap();
assert_eq!(fs::read(store_path.join(raw_relpath)).unwrap(), body.as_bytes());
let _ = fs::remove_dir_all(&base_path);
}
#[test]
fn test_text_capture_hides_synthetic_url_but_reuses_source_identity() {
let base_path = env::temp_dir().join(format!(
"archivr-text-identity-test-{}",
Local::now().format("%Y%m%d%H%M%S%3f")
));
let _ = fs::remove_dir_all(&base_path);
fs::create_dir_all(&base_path).unwrap();
let store_path = base_path.join("store");
let archive_path = base_path.join(".archivr");
archive::initialize_store_directories(&store_path).unwrap();
fs::create_dir_all(&archive_path).unwrap();
fs::write(archive_path.join("name"), "test-archive").unwrap();
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
let archive_paths = ArchivePaths {
archive_path: archive_path.clone(),
store_path,
name: "test-archive".to_string(),
};
let body = "Same body, same text identity.";
perform_text_capture(&archive_paths, "First title", body, "text/plain", None).unwrap();
perform_text_capture(&archive_paths, "Second title", body, "text/plain", None).unwrap();
let conn = database::open_or_initialize(&archive_path).unwrap();
let default_coll_id = database::ensure_default_collection(&conn).unwrap();
let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
assert_eq!(entries.len(), 2);
assert!(entries.iter().all(|entry| entry.original_url.is_none()));
let expected_locator = format!("text:{}", crate::hash::hash_bytes(body.as_bytes()));
let (canonical_url, normalized_locator): (Option<String>, String) = conn
.query_row(
"SELECT canonical_url, normalized_locator
FROM source_identities
WHERE source_kind = 'text' AND normalized_locator = ?1",
[&expected_locator],
|row| Ok((row.get(0)?, row.get(1)?)),
)
.unwrap();
assert_eq!(canonical_url, None);
assert_eq!(normalized_locator, expected_locator);
let source_identity_count: i64 = conn
.query_row(
"SELECT COUNT(*) FROM source_identities WHERE source_kind = 'text'",
[],
|row| row.get(0),
)
.unwrap();
assert_eq!(source_identity_count, 1);
let _ = fs::remove_dir_all(&base_path);
}
#[test] #[test]
fn test_text_capture_rejects_empty_title() { fn test_text_capture_rejects_empty_title() {
let base_path = env::temp_dir().join(format!( let base_path = env::temp_dir().join(format!(

View file

@ -1419,7 +1419,7 @@ async fn capture_text_handler(
// Spawn background text capture. // Spawn background text capture.
let title = body.title.trim().to_string(); let title = body.title.trim().to_string();
let text_body = body.body.trim().to_string(); let text_body = body.body;
let mime_str = mime.to_string(); let mime_str = mime.to_string();
let job_uid_bg = job_uid.clone(); let job_uid_bg = job_uid.clone();
let archive_path = mounted.archive_path.clone(); let archive_path = mounted.archive_path.clone();
@ -4241,6 +4241,71 @@ mod tests {
assert_eq!(json["status"], "pending"); assert_eq!(json["status"], "pending");
} }
#[tokio::test]
async fn text_capture_post_preserves_intentional_whitespace() {
let dir = tempfile::tempdir().unwrap();
let (registry, archive_path, auth_path) = make_test_registry(&dir);
let session_cookie = make_test_session(&auth_path);
let text_body = " \n# Heading\n\nContent with a final newline\n\t ";
let response = app(registry, auth_path)
.oneshot(
Request::builder()
.method("POST")
.uri("/api/archives/test/captures/text")
.header("content-type", "application/json")
.header("cookie", &session_cookie)
.body(json_body(&serde_json::json!({
"title": "Whitespace Note",
"body": text_body,
"mime": "text/markdown"
})))
.unwrap(),
)
.await
.unwrap();
assert_eq!(response.status(), StatusCode::ACCEPTED);
let response_body = axum::body::to_bytes(response.into_body(), usize::MAX)
.await
.unwrap();
let job_uid = serde_json::from_slice::<serde_json::Value>(&response_body).unwrap()["job_uid"]
.as_str()
.unwrap()
.to_string();
let status = tokio::time::timeout(std::time::Duration::from_secs(2), async {
loop {
let conn = archivr_core::database::open_or_initialize(&archive_path).unwrap();
let job = archivr_core::database::get_capture_job(&conn, &job_uid)
.unwrap()
.unwrap();
if job.status != "pending" && job.status != "running" {
break job.status;
}
tokio::time::sleep(std::time::Duration::from_millis(10)).await;
}
})
.await
.expect("text capture job should finish");
assert_eq!(status, "completed");
let archive_paths = archivr_core::archive::read_archive_paths(&archive_path).unwrap();
let conn = archivr_core::database::open_or_initialize(&archive_path).unwrap();
let raw_relpath: String = conn
.query_row(
"SELECT b.raw_relpath
FROM entry_artifacts ea
JOIN blobs b ON b.id = ea.blob_id
WHERE ea.artifact_role = 'primary_media'",
[],
|row| row.get(0),
)
.unwrap();
assert_eq!(
std::fs::read(archive_paths.store_path.join(raw_relpath)).unwrap(),
text_body.as_bytes()
);
}
#[tokio::test] #[tokio::test]
async fn text_capture_rejects_empty_title() { async fn text_capture_rejects_empty_title() {
let dir = tempfile::tempdir().unwrap(); let dir = tempfile::tempdir().unwrap();