1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-10-09 12:55:00 +02:00

Merge branch 'x-article-summary' into integration-all-three

This commit is contained in:
archivr-qa 2026-08-23 20:53:15 +02:00
commit 6c6de347d1
No known key found for this signature in database
3 changed files with 210 additions and 9 deletions

1
Cargo.lock generated
View file

@ -1543,6 +1543,7 @@ version = "1.0.150"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9"
dependencies = [
"indexmap",
"itoa",
"memchr",
"serde",

View file

@ -19,7 +19,7 @@ hex = "0.4.3"
regex = "1.12.2"
rusqlite = { version = "0.32.1", features = ["bundled"] }
serde = { version = "1.0.228", features = ["derive"] }
serde_json = "1.0.132"
serde_json = { version = "1.0.132", features = ["preserve_order"] }
sha3 = "0.10.8"
tempfile = "3.13.0"
tokio = { version = "1.41.1", features = ["macros", "rt-multi-thread", "net", "fs", "io-util"] }

View file

@ -682,16 +682,64 @@ fn decode_entities(s: &str) -> String {
/// Pulls the human text out of a scraped tweet JSON payload, tolerating both
/// the flat shape and a `{ "tweet": { … } }` wrapper, and appending any thread
/// entries so a self-reply chain summarizes as one piece.
pub fn extract_tweet_text(json: &serde_json::Value) -> Option<String> {
fn one(v: &serde_json::Value) -> Option<String> {
for key in ["full_text", "text", "content", "body"] {
if let Some(s) = v.get(key).and_then(|x| x.as_str()) {
if !s.trim().is_empty() {
return Some(s.to_string());
}
fn nonempty_string(v: &serde_json::Value, key: &str) -> Option<String> {
v.get(key)
.and_then(|value| value.as_str())
.filter(|value| !value.trim().is_empty())
.map(str::to_string)
}
fn flatten_article_blocks(v: &serde_json::Value, out: &mut Vec<String>) {
const TEXT_KEYS: &[&str] = &["text", "plain_text", "content", "body", "title", "heading"];
match v {
serde_json::Value::Array(items) => {
for item in items {
flatten_article_blocks(item, out);
}
}
None
serde_json::Value::Object(fields) => {
for (key, value) in fields {
if TEXT_KEYS.contains(&key.as_str()) {
if let Some(text) = value.as_str().filter(|text| !text.trim().is_empty()) {
out.push(text.to_string());
}
}
flatten_article_blocks(value, out);
}
}
_ => {}
}
}
fn article_text(status: &serde_json::Value) -> Option<String> {
let article = status.get("article")?;
article.as_object()?;
let title = nonempty_string(article, "title");
let body = nonempty_string(article, "plain_text")
.or_else(|| {
let mut blocks = Vec::new();
if let Some(value) = article.get("blocks") {
flatten_article_blocks(value, &mut blocks);
}
(!blocks.is_empty()).then(|| blocks.join("\n\n"))
})
.or_else(|| nonempty_string(article, "preview_text"))
.or_else(|| nonempty_string(article, "summary_text"))?;
Some(match title.as_deref() {
Some(title) => format!("{title}\n\n{body}"),
None => body,
})
}
pub fn extract_tweet_text(json: &serde_json::Value) -> Option<String> {
fn one(v: &serde_json::Value) -> Option<String> {
article_text(v).or_else(|| {
["full_text", "text", "content", "body"]
.into_iter()
.find_map(|key| nonempty_string(v, key))
})
}
let root = json.get("tweet").unwrap_or(json);
let mut parts: Vec<String> = Vec::new();
@ -1199,6 +1247,158 @@ mod tests {
assert!(extract_tweet_text(&serde_json::json!({ "id": 1 })).is_none());
}
#[test]
fn extract_tweet_text_prefers_x_article_plain_text_over_tco_body() {
let tweet = serde_json::json!({
"full_text": "https://t.co/article",
"article": { "title": "Skin guide", "plain_text": "Use sunscreen daily." }
});
assert_eq!(
extract_tweet_text(&tweet).as_deref(),
Some("Skin guide\n\nUse sunscreen daily.")
);
}
#[test]
fn extract_tweet_text_uses_article_blocks_when_plain_text_is_empty() {
let tweet = serde_json::json!({"article": {
"title": "Blocks", "plain_text": " ",
"blocks": [
{"text": "First", "id": "ignored", "media_url": "https://example.test/image"},
{"children": [{"text": "Second", "enabled": true}]}
]
}});
assert_eq!(
extract_tweet_text(&tweet).as_deref(),
Some("Blocks\n\nFirst\n\nSecond")
);
}
#[test]
fn extract_tweet_text_keeps_article_block_object_field_order() {
let tweet: serde_json::Value = serde_json::from_str(
r#"{
"article": {
"title": "Ordered block",
"blocks": [{"heading": "Opening", "content": "Body copy"}]
}
}"#,
)
.unwrap();
assert_eq!(
extract_tweet_text(&tweet).as_deref(),
Some("Ordered block\n\nOpening\n\nBody copy")
);
}
#[test]
fn extract_tweet_text_falls_back_from_article_preview_to_summary_then_tweet_body() {
let preview = serde_json::json!({
"full_text": "https://t.co/fallback",
"article": { "title": "Preview", "preview_text": "Preview copy", "summary_text": "Later" }
});
assert_eq!(
extract_tweet_text(&preview).as_deref(),
Some("Preview\n\nPreview copy")
);
let summary = serde_json::json!({
"full_text": "https://t.co/fallback",
"article": { "title": "Summary", "summary_text": "Summary copy" }
});
assert_eq!(
extract_tweet_text(&summary).as_deref(),
Some("Summary\n\nSummary copy")
);
let empty_article = serde_json::json!({
"full_text": "https://t.co/fallback",
"article": { "title": "Only a title", "blocks": [{"id": "not text"}] }
});
assert_eq!(
extract_tweet_text(&empty_article).as_deref(),
Some("https://t.co/fallback")
);
}
#[test]
fn build_summary_input_joins_article_backed_tweet_artifacts() {
let temp = tempfile::tempdir().unwrap();
let paths = crate::archive::initialize_archive(
temp.path(),
&temp.path().join("store"),
"Test archive",
false,
)
.unwrap();
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
let user_id = database::ensure_default_user(&conn).unwrap();
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
let source_id = database::upsert_source_identity(
&conn,
"twitter",
"tweet_thread",
Some("thread-1"),
Some("https://x.com/example/status/1"),
"x:thread-1",
)
.unwrap();
let entry = database::create_archived_entry(
&conn,
&database::NewEntry {
source_identity_id: source_id,
archive_run_id: run.id,
parent_entry_id: None,
root_entry_id: None,
created_by_user_id: user_id,
owned_by_user_id: user_id,
source_kind: "twitter".to_string(),
entity_kind: "tweet_thread".to_string(),
title: Some("Article thread".to_string()),
visibility: "private".to_string(),
representation_kind: "tweet_thread".to_string(),
source_metadata_json: "{}".to_string(),
display_metadata_json: None,
},
)
.unwrap();
for (ordinal, (relpath, body)) in [
("raw_tweets/article-one.json", "First article body."),
("raw_tweets/article-two.json", "Second article body."),
]
.into_iter()
.enumerate()
{
let path = paths.store_path.join(relpath);
std::fs::write(
&path,
serde_json::json!({
"full_text": format!("https://t.co/{ordinal}"),
"article": { "title": format!("Article {}", ordinal + 1), "plain_text": body }
})
.to_string(),
)
.unwrap();
database::add_entry_artifact(
&conn,
&database::NewArtifact {
entry_id: entry.id,
artifact_role: "raw_tweet_json".to_string(),
storage_area: "raw_tweets".to_string(),
relpath: relpath.to_string(),
blob_id: None,
logical_path: None,
metadata_json: None,
},
)
.unwrap();
}
let input = build_summary_input(&paths, &entry.entry_uid).unwrap();
assert!(input.request.content.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body."));
}
// ── Output normalization ───────────────────────────────────────────────
#[test]