mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 12:55:00 +02:00
fix: summarize X Article text
This commit is contained in:
parent
7a10fedaf2
commit
b17db8d04a
1 changed files with 191 additions and 8 deletions
|
|
@ -682,16 +682,64 @@ fn decode_entities(s: &str) -> String {
|
|||
/// Pulls the human text out of a scraped tweet JSON payload, tolerating both
|
||||
/// the flat shape and a `{ "tweet": { … } }` wrapper, and appending any thread
|
||||
/// entries so a self-reply chain summarizes as one piece.
|
||||
fn nonempty_string(v: &serde_json::Value, key: &str) -> Option<String> {
|
||||
v.get(key)
|
||||
.and_then(|value| value.as_str())
|
||||
.filter(|value| !value.trim().is_empty())
|
||||
.map(str::to_string)
|
||||
}
|
||||
|
||||
fn flatten_article_blocks(v: &serde_json::Value, out: &mut Vec<String>) {
|
||||
const TEXT_KEYS: &[&str] = &["text", "plain_text", "content", "body", "title", "heading"];
|
||||
|
||||
match v {
|
||||
serde_json::Value::Array(items) => {
|
||||
for item in items {
|
||||
flatten_article_blocks(item, out);
|
||||
}
|
||||
}
|
||||
serde_json::Value::Object(fields) => {
|
||||
for (key, value) in fields {
|
||||
if TEXT_KEYS.contains(&key.as_str()) {
|
||||
if let Some(text) = value.as_str().filter(|text| !text.trim().is_empty()) {
|
||||
out.push(text.to_string());
|
||||
}
|
||||
}
|
||||
flatten_article_blocks(value, out);
|
||||
}
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
fn article_text(status: &serde_json::Value) -> Option<String> {
|
||||
let article = status.get("article")?;
|
||||
article.as_object()?;
|
||||
let title = nonempty_string(article, "title");
|
||||
let body = nonempty_string(article, "plain_text")
|
||||
.or_else(|| {
|
||||
let mut blocks = Vec::new();
|
||||
if let Some(value) = article.get("blocks") {
|
||||
flatten_article_blocks(value, &mut blocks);
|
||||
}
|
||||
(!blocks.is_empty()).then(|| blocks.join("\n\n"))
|
||||
})
|
||||
.or_else(|| nonempty_string(article, "preview_text"))
|
||||
.or_else(|| nonempty_string(article, "summary_text"))?;
|
||||
|
||||
Some(match title.as_deref() {
|
||||
Some(title) => format!("{title}\n\n{body}"),
|
||||
None => body,
|
||||
})
|
||||
}
|
||||
|
||||
pub fn extract_tweet_text(json: &serde_json::Value) -> Option<String> {
|
||||
fn one(v: &serde_json::Value) -> Option<String> {
|
||||
for key in ["full_text", "text", "content", "body"] {
|
||||
if let Some(s) = v.get(key).and_then(|x| x.as_str()) {
|
||||
if !s.trim().is_empty() {
|
||||
return Some(s.to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
None
|
||||
article_text(v).or_else(|| {
|
||||
["full_text", "text", "content", "body"]
|
||||
.into_iter()
|
||||
.find_map(|key| nonempty_string(v, key))
|
||||
})
|
||||
}
|
||||
let root = json.get("tweet").unwrap_or(json);
|
||||
let mut parts: Vec<String> = Vec::new();
|
||||
|
|
@ -1199,6 +1247,141 @@ mod tests {
|
|||
assert!(extract_tweet_text(&serde_json::json!({ "id": 1 })).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extract_tweet_text_prefers_x_article_plain_text_over_tco_body() {
|
||||
let tweet = serde_json::json!({
|
||||
"full_text": "https://t.co/article",
|
||||
"article": { "title": "Skin guide", "plain_text": "Use sunscreen daily." }
|
||||
});
|
||||
assert_eq!(
|
||||
extract_tweet_text(&tweet).as_deref(),
|
||||
Some("Skin guide\n\nUse sunscreen daily.")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extract_tweet_text_uses_article_blocks_when_plain_text_is_empty() {
|
||||
let tweet = serde_json::json!({"article": {
|
||||
"title": "Blocks", "plain_text": " ",
|
||||
"blocks": [
|
||||
{"text": "First", "id": "ignored", "media_url": "https://example.test/image"},
|
||||
{"children": [{"text": "Second", "enabled": true}]}
|
||||
]
|
||||
}});
|
||||
assert_eq!(
|
||||
extract_tweet_text(&tweet).as_deref(),
|
||||
Some("Blocks\n\nFirst\n\nSecond")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extract_tweet_text_falls_back_from_article_preview_to_summary_then_tweet_body() {
|
||||
let preview = serde_json::json!({
|
||||
"full_text": "https://t.co/fallback",
|
||||
"article": { "title": "Preview", "preview_text": "Preview copy", "summary_text": "Later" }
|
||||
});
|
||||
assert_eq!(
|
||||
extract_tweet_text(&preview).as_deref(),
|
||||
Some("Preview\n\nPreview copy")
|
||||
);
|
||||
|
||||
let summary = serde_json::json!({
|
||||
"full_text": "https://t.co/fallback",
|
||||
"article": { "title": "Summary", "summary_text": "Summary copy" }
|
||||
});
|
||||
assert_eq!(
|
||||
extract_tweet_text(&summary).as_deref(),
|
||||
Some("Summary\n\nSummary copy")
|
||||
);
|
||||
|
||||
let empty_article = serde_json::json!({
|
||||
"full_text": "https://t.co/fallback",
|
||||
"article": { "title": "Only a title", "blocks": [{"id": "not text"}] }
|
||||
});
|
||||
assert_eq!(
|
||||
extract_tweet_text(&empty_article).as_deref(),
|
||||
Some("https://t.co/fallback")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn build_summary_input_joins_article_backed_tweet_artifacts() {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let paths = crate::archive::initialize_archive(
|
||||
temp.path(),
|
||||
&temp.path().join("store"),
|
||||
"Test archive",
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
let user_id = database::ensure_default_user(&conn).unwrap();
|
||||
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
|
||||
let source_id = database::upsert_source_identity(
|
||||
&conn,
|
||||
"twitter",
|
||||
"tweet_thread",
|
||||
Some("thread-1"),
|
||||
Some("https://x.com/example/status/1"),
|
||||
"x:thread-1",
|
||||
)
|
||||
.unwrap();
|
||||
let entry = database::create_archived_entry(
|
||||
&conn,
|
||||
&database::NewEntry {
|
||||
source_identity_id: source_id,
|
||||
archive_run_id: run.id,
|
||||
parent_entry_id: None,
|
||||
root_entry_id: None,
|
||||
created_by_user_id: user_id,
|
||||
owned_by_user_id: user_id,
|
||||
source_kind: "twitter".to_string(),
|
||||
entity_kind: "tweet_thread".to_string(),
|
||||
title: Some("Article thread".to_string()),
|
||||
visibility: "private".to_string(),
|
||||
representation_kind: "tweet_thread".to_string(),
|
||||
source_metadata_json: "{}".to_string(),
|
||||
display_metadata_json: None,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
for (ordinal, (relpath, body)) in [
|
||||
("raw_tweets/article-one.json", "First article body."),
|
||||
("raw_tweets/article-two.json", "Second article body."),
|
||||
]
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
{
|
||||
let path = paths.store_path.join(relpath);
|
||||
std::fs::write(
|
||||
&path,
|
||||
serde_json::json!({
|
||||
"full_text": format!("https://t.co/{ordinal}"),
|
||||
"article": { "title": format!("Article {}", ordinal + 1), "plain_text": body }
|
||||
})
|
||||
.to_string(),
|
||||
)
|
||||
.unwrap();
|
||||
database::add_entry_artifact(
|
||||
&conn,
|
||||
&database::NewArtifact {
|
||||
entry_id: entry.id,
|
||||
artifact_role: "raw_tweet_json".to_string(),
|
||||
storage_area: "raw_tweets".to_string(),
|
||||
relpath: relpath.to_string(),
|
||||
blob_id: None,
|
||||
logical_path: None,
|
||||
metadata_json: None,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let input = build_summary_input(&paths, &entry.entry_uid).unwrap();
|
||||
assert!(input.request.content.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body."));
|
||||
}
|
||||
|
||||
// ── Output normalization ───────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue