mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 12:55:00 +02:00
fix: preserve text capture bytes and hide synthetic URL
This commit is contained in:
parent
492b1a3168
commit
771c64a811
2 changed files with 173 additions and 4 deletions
|
|
@ -1951,8 +1951,7 @@ pub fn perform_text_capture(
|
||||||
}
|
}
|
||||||
|
|
||||||
// Validate body
|
// Validate body
|
||||||
let body = body.trim();
|
if body.trim().is_empty() {
|
||||||
if body.is_empty() {
|
|
||||||
anyhow::bail!("body must not be empty");
|
anyhow::bail!("body must not be empty");
|
||||||
}
|
}
|
||||||
if body.len() > 2 * 1024 * 1024 {
|
if body.len() > 2 * 1024 * 1024 {
|
||||||
|
|
@ -2024,7 +2023,7 @@ pub fn perform_text_capture(
|
||||||
source_kind,
|
source_kind,
|
||||||
entity_kind,
|
entity_kind,
|
||||||
None,
|
None,
|
||||||
Some(&canonical_locator),
|
None,
|
||||||
&canonical_locator,
|
&canonical_locator,
|
||||||
)?;
|
)?;
|
||||||
|
|
||||||
|
|
@ -2887,6 +2886,111 @@ mod tests {
|
||||||
let _ = fs::remove_dir_all(&base_path);
|
let _ = fs::remove_dir_all(&base_path);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_text_capture_preserves_intentional_whitespace_in_stored_artifact() {
|
||||||
|
let base_path = env::temp_dir().join(format!(
|
||||||
|
"archivr-text-whitespace-test-{}",
|
||||||
|
Local::now().format("%Y%m%d%H%M%S%3f")
|
||||||
|
));
|
||||||
|
let _ = fs::remove_dir_all(&base_path);
|
||||||
|
fs::create_dir_all(&base_path).unwrap();
|
||||||
|
|
||||||
|
let store_path = base_path.join("store");
|
||||||
|
let archive_path = base_path.join(".archivr");
|
||||||
|
archive::initialize_store_directories(&store_path).unwrap();
|
||||||
|
fs::create_dir_all(&archive_path).unwrap();
|
||||||
|
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||||
|
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||||
|
|
||||||
|
let archive_paths = ArchivePaths {
|
||||||
|
archive_path: archive_path.clone(),
|
||||||
|
store_path: store_path.clone(),
|
||||||
|
name: "test-archive".to_string(),
|
||||||
|
};
|
||||||
|
let body = " \n# Heading\n\nContent with a final newline\n\t ";
|
||||||
|
|
||||||
|
perform_text_capture(
|
||||||
|
&archive_paths,
|
||||||
|
"Whitespace Note",
|
||||||
|
body,
|
||||||
|
"text/markdown",
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let conn = database::open_or_initialize(&archive_path).unwrap();
|
||||||
|
let raw_relpath: String = conn
|
||||||
|
.query_row(
|
||||||
|
"SELECT b.raw_relpath
|
||||||
|
FROM entry_artifacts ea
|
||||||
|
JOIN blobs b ON b.id = ea.blob_id
|
||||||
|
WHERE ea.artifact_role = 'primary_media'",
|
||||||
|
[],
|
||||||
|
|row| row.get(0),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(fs::read(store_path.join(raw_relpath)).unwrap(), body.as_bytes());
|
||||||
|
|
||||||
|
let _ = fs::remove_dir_all(&base_path);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_text_capture_hides_synthetic_url_but_reuses_source_identity() {
|
||||||
|
let base_path = env::temp_dir().join(format!(
|
||||||
|
"archivr-text-identity-test-{}",
|
||||||
|
Local::now().format("%Y%m%d%H%M%S%3f")
|
||||||
|
));
|
||||||
|
let _ = fs::remove_dir_all(&base_path);
|
||||||
|
fs::create_dir_all(&base_path).unwrap();
|
||||||
|
|
||||||
|
let store_path = base_path.join("store");
|
||||||
|
let archive_path = base_path.join(".archivr");
|
||||||
|
archive::initialize_store_directories(&store_path).unwrap();
|
||||||
|
fs::create_dir_all(&archive_path).unwrap();
|
||||||
|
fs::write(archive_path.join("name"), "test-archive").unwrap();
|
||||||
|
fs::write(archive_path.join("store_path"), store_path.to_str().unwrap()).unwrap();
|
||||||
|
|
||||||
|
let archive_paths = ArchivePaths {
|
||||||
|
archive_path: archive_path.clone(),
|
||||||
|
store_path,
|
||||||
|
name: "test-archive".to_string(),
|
||||||
|
};
|
||||||
|
let body = "Same body, same text identity.";
|
||||||
|
|
||||||
|
perform_text_capture(&archive_paths, "First title", body, "text/plain", None).unwrap();
|
||||||
|
perform_text_capture(&archive_paths, "Second title", body, "text/plain", None).unwrap();
|
||||||
|
|
||||||
|
let conn = database::open_or_initialize(&archive_path).unwrap();
|
||||||
|
let default_coll_id = database::ensure_default_collection(&conn).unwrap();
|
||||||
|
let entries = archive::list_entries_for_collection(&conn, default_coll_id, 0xFFFFFFFF).unwrap();
|
||||||
|
assert_eq!(entries.len(), 2);
|
||||||
|
assert!(entries.iter().all(|entry| entry.original_url.is_none()));
|
||||||
|
|
||||||
|
let expected_locator = format!("text:{}", crate::hash::hash_bytes(body.as_bytes()));
|
||||||
|
let (canonical_url, normalized_locator): (Option<String>, String) = conn
|
||||||
|
.query_row(
|
||||||
|
"SELECT canonical_url, normalized_locator
|
||||||
|
FROM source_identities
|
||||||
|
WHERE source_kind = 'text' AND normalized_locator = ?1",
|
||||||
|
[&expected_locator],
|
||||||
|
|row| Ok((row.get(0)?, row.get(1)?)),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(canonical_url, None);
|
||||||
|
assert_eq!(normalized_locator, expected_locator);
|
||||||
|
|
||||||
|
let source_identity_count: i64 = conn
|
||||||
|
.query_row(
|
||||||
|
"SELECT COUNT(*) FROM source_identities WHERE source_kind = 'text'",
|
||||||
|
[],
|
||||||
|
|row| row.get(0),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(source_identity_count, 1);
|
||||||
|
|
||||||
|
let _ = fs::remove_dir_all(&base_path);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_text_capture_rejects_empty_title() {
|
fn test_text_capture_rejects_empty_title() {
|
||||||
let base_path = env::temp_dir().join(format!(
|
let base_path = env::temp_dir().join(format!(
|
||||||
|
|
|
||||||
|
|
@ -1419,7 +1419,7 @@ async fn capture_text_handler(
|
||||||
|
|
||||||
// Spawn background text capture.
|
// Spawn background text capture.
|
||||||
let title = body.title.trim().to_string();
|
let title = body.title.trim().to_string();
|
||||||
let text_body = body.body.trim().to_string();
|
let text_body = body.body;
|
||||||
let mime_str = mime.to_string();
|
let mime_str = mime.to_string();
|
||||||
let job_uid_bg = job_uid.clone();
|
let job_uid_bg = job_uid.clone();
|
||||||
let archive_path = mounted.archive_path.clone();
|
let archive_path = mounted.archive_path.clone();
|
||||||
|
|
@ -4241,6 +4241,71 @@ mod tests {
|
||||||
assert_eq!(json["status"], "pending");
|
assert_eq!(json["status"], "pending");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn text_capture_post_preserves_intentional_whitespace() {
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let (registry, archive_path, auth_path) = make_test_registry(&dir);
|
||||||
|
let session_cookie = make_test_session(&auth_path);
|
||||||
|
let text_body = " \n# Heading\n\nContent with a final newline\n\t ";
|
||||||
|
let response = app(registry, auth_path)
|
||||||
|
.oneshot(
|
||||||
|
Request::builder()
|
||||||
|
.method("POST")
|
||||||
|
.uri("/api/archives/test/captures/text")
|
||||||
|
.header("content-type", "application/json")
|
||||||
|
.header("cookie", &session_cookie)
|
||||||
|
.body(json_body(&serde_json::json!({
|
||||||
|
"title": "Whitespace Note",
|
||||||
|
"body": text_body,
|
||||||
|
"mime": "text/markdown"
|
||||||
|
})))
|
||||||
|
.unwrap(),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(response.status(), StatusCode::ACCEPTED);
|
||||||
|
let response_body = axum::body::to_bytes(response.into_body(), usize::MAX)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let job_uid = serde_json::from_slice::<serde_json::Value>(&response_body).unwrap()["job_uid"]
|
||||||
|
.as_str()
|
||||||
|
.unwrap()
|
||||||
|
.to_string();
|
||||||
|
|
||||||
|
let status = tokio::time::timeout(std::time::Duration::from_secs(2), async {
|
||||||
|
loop {
|
||||||
|
let conn = archivr_core::database::open_or_initialize(&archive_path).unwrap();
|
||||||
|
let job = archivr_core::database::get_capture_job(&conn, &job_uid)
|
||||||
|
.unwrap()
|
||||||
|
.unwrap();
|
||||||
|
if job.status != "pending" && job.status != "running" {
|
||||||
|
break job.status;
|
||||||
|
}
|
||||||
|
tokio::time::sleep(std::time::Duration::from_millis(10)).await;
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.expect("text capture job should finish");
|
||||||
|
assert_eq!(status, "completed");
|
||||||
|
|
||||||
|
let archive_paths = archivr_core::archive::read_archive_paths(&archive_path).unwrap();
|
||||||
|
let conn = archivr_core::database::open_or_initialize(&archive_path).unwrap();
|
||||||
|
let raw_relpath: String = conn
|
||||||
|
.query_row(
|
||||||
|
"SELECT b.raw_relpath
|
||||||
|
FROM entry_artifacts ea
|
||||||
|
JOIN blobs b ON b.id = ea.blob_id
|
||||||
|
WHERE ea.artifact_role = 'primary_media'",
|
||||||
|
[],
|
||||||
|
|row| row.get(0),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
std::fs::read(archive_paths.store_path.join(raw_relpath)).unwrap(),
|
||||||
|
text_body.as_bytes()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn text_capture_rejects_empty_title() {
|
async fn text_capture_rejects_empty_title() {
|
||||||
let dir = tempfile::tempdir().unwrap();
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue