mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 12:55:00 +02:00
Merge branch 'x-article-summary' into integration-all-three
This commit is contained in:
commit
6c6de347d1
3 changed files with 210 additions and 9 deletions
1
Cargo.lock
generated
1
Cargo.lock
generated
|
|
@ -1543,6 +1543,7 @@ version = "1.0.150"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9"
|
checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
|
"indexmap",
|
||||||
"itoa",
|
"itoa",
|
||||||
"memchr",
|
"memchr",
|
||||||
"serde",
|
"serde",
|
||||||
|
|
|
||||||
|
|
@ -19,7 +19,7 @@ hex = "0.4.3"
|
||||||
regex = "1.12.2"
|
regex = "1.12.2"
|
||||||
rusqlite = { version = "0.32.1", features = ["bundled"] }
|
rusqlite = { version = "0.32.1", features = ["bundled"] }
|
||||||
serde = { version = "1.0.228", features = ["derive"] }
|
serde = { version = "1.0.228", features = ["derive"] }
|
||||||
serde_json = "1.0.132"
|
serde_json = { version = "1.0.132", features = ["preserve_order"] }
|
||||||
sha3 = "0.10.8"
|
sha3 = "0.10.8"
|
||||||
tempfile = "3.13.0"
|
tempfile = "3.13.0"
|
||||||
tokio = { version = "1.41.1", features = ["macros", "rt-multi-thread", "net", "fs", "io-util"] }
|
tokio = { version = "1.41.1", features = ["macros", "rt-multi-thread", "net", "fs", "io-util"] }
|
||||||
|
|
|
||||||
|
|
@ -682,16 +682,64 @@ fn decode_entities(s: &str) -> String {
|
||||||
/// Pulls the human text out of a scraped tweet JSON payload, tolerating both
|
/// Pulls the human text out of a scraped tweet JSON payload, tolerating both
|
||||||
/// the flat shape and a `{ "tweet": { … } }` wrapper, and appending any thread
|
/// the flat shape and a `{ "tweet": { … } }` wrapper, and appending any thread
|
||||||
/// entries so a self-reply chain summarizes as one piece.
|
/// entries so a self-reply chain summarizes as one piece.
|
||||||
pub fn extract_tweet_text(json: &serde_json::Value) -> Option<String> {
|
fn nonempty_string(v: &serde_json::Value, key: &str) -> Option<String> {
|
||||||
fn one(v: &serde_json::Value) -> Option<String> {
|
v.get(key)
|
||||||
for key in ["full_text", "text", "content", "body"] {
|
.and_then(|value| value.as_str())
|
||||||
if let Some(s) = v.get(key).and_then(|x| x.as_str()) {
|
.filter(|value| !value.trim().is_empty())
|
||||||
if !s.trim().is_empty() {
|
.map(str::to_string)
|
||||||
return Some(s.to_string());
|
}
|
||||||
}
|
|
||||||
|
fn flatten_article_blocks(v: &serde_json::Value, out: &mut Vec<String>) {
|
||||||
|
const TEXT_KEYS: &[&str] = &["text", "plain_text", "content", "body", "title", "heading"];
|
||||||
|
|
||||||
|
match v {
|
||||||
|
serde_json::Value::Array(items) => {
|
||||||
|
for item in items {
|
||||||
|
flatten_article_blocks(item, out);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
None
|
serde_json::Value::Object(fields) => {
|
||||||
|
for (key, value) in fields {
|
||||||
|
if TEXT_KEYS.contains(&key.as_str()) {
|
||||||
|
if let Some(text) = value.as_str().filter(|text| !text.trim().is_empty()) {
|
||||||
|
out.push(text.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
flatten_article_blocks(value, out);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn article_text(status: &serde_json::Value) -> Option<String> {
|
||||||
|
let article = status.get("article")?;
|
||||||
|
article.as_object()?;
|
||||||
|
let title = nonempty_string(article, "title");
|
||||||
|
let body = nonempty_string(article, "plain_text")
|
||||||
|
.or_else(|| {
|
||||||
|
let mut blocks = Vec::new();
|
||||||
|
if let Some(value) = article.get("blocks") {
|
||||||
|
flatten_article_blocks(value, &mut blocks);
|
||||||
|
}
|
||||||
|
(!blocks.is_empty()).then(|| blocks.join("\n\n"))
|
||||||
|
})
|
||||||
|
.or_else(|| nonempty_string(article, "preview_text"))
|
||||||
|
.or_else(|| nonempty_string(article, "summary_text"))?;
|
||||||
|
|
||||||
|
Some(match title.as_deref() {
|
||||||
|
Some(title) => format!("{title}\n\n{body}"),
|
||||||
|
None => body,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn extract_tweet_text(json: &serde_json::Value) -> Option<String> {
|
||||||
|
fn one(v: &serde_json::Value) -> Option<String> {
|
||||||
|
article_text(v).or_else(|| {
|
||||||
|
["full_text", "text", "content", "body"]
|
||||||
|
.into_iter()
|
||||||
|
.find_map(|key| nonempty_string(v, key))
|
||||||
|
})
|
||||||
}
|
}
|
||||||
let root = json.get("tweet").unwrap_or(json);
|
let root = json.get("tweet").unwrap_or(json);
|
||||||
let mut parts: Vec<String> = Vec::new();
|
let mut parts: Vec<String> = Vec::new();
|
||||||
|
|
@ -1199,6 +1247,158 @@ mod tests {
|
||||||
assert!(extract_tweet_text(&serde_json::json!({ "id": 1 })).is_none());
|
assert!(extract_tweet_text(&serde_json::json!({ "id": 1 })).is_none());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extract_tweet_text_prefers_x_article_plain_text_over_tco_body() {
|
||||||
|
let tweet = serde_json::json!({
|
||||||
|
"full_text": "https://t.co/article",
|
||||||
|
"article": { "title": "Skin guide", "plain_text": "Use sunscreen daily." }
|
||||||
|
});
|
||||||
|
assert_eq!(
|
||||||
|
extract_tweet_text(&tweet).as_deref(),
|
||||||
|
Some("Skin guide\n\nUse sunscreen daily.")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extract_tweet_text_uses_article_blocks_when_plain_text_is_empty() {
|
||||||
|
let tweet = serde_json::json!({"article": {
|
||||||
|
"title": "Blocks", "plain_text": " ",
|
||||||
|
"blocks": [
|
||||||
|
{"text": "First", "id": "ignored", "media_url": "https://example.test/image"},
|
||||||
|
{"children": [{"text": "Second", "enabled": true}]}
|
||||||
|
]
|
||||||
|
}});
|
||||||
|
assert_eq!(
|
||||||
|
extract_tweet_text(&tweet).as_deref(),
|
||||||
|
Some("Blocks\n\nFirst\n\nSecond")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extract_tweet_text_keeps_article_block_object_field_order() {
|
||||||
|
let tweet: serde_json::Value = serde_json::from_str(
|
||||||
|
r#"{
|
||||||
|
"article": {
|
||||||
|
"title": "Ordered block",
|
||||||
|
"blocks": [{"heading": "Opening", "content": "Body copy"}]
|
||||||
|
}
|
||||||
|
}"#,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
extract_tweet_text(&tweet).as_deref(),
|
||||||
|
Some("Ordered block\n\nOpening\n\nBody copy")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extract_tweet_text_falls_back_from_article_preview_to_summary_then_tweet_body() {
|
||||||
|
let preview = serde_json::json!({
|
||||||
|
"full_text": "https://t.co/fallback",
|
||||||
|
"article": { "title": "Preview", "preview_text": "Preview copy", "summary_text": "Later" }
|
||||||
|
});
|
||||||
|
assert_eq!(
|
||||||
|
extract_tweet_text(&preview).as_deref(),
|
||||||
|
Some("Preview\n\nPreview copy")
|
||||||
|
);
|
||||||
|
|
||||||
|
let summary = serde_json::json!({
|
||||||
|
"full_text": "https://t.co/fallback",
|
||||||
|
"article": { "title": "Summary", "summary_text": "Summary copy" }
|
||||||
|
});
|
||||||
|
assert_eq!(
|
||||||
|
extract_tweet_text(&summary).as_deref(),
|
||||||
|
Some("Summary\n\nSummary copy")
|
||||||
|
);
|
||||||
|
|
||||||
|
let empty_article = serde_json::json!({
|
||||||
|
"full_text": "https://t.co/fallback",
|
||||||
|
"article": { "title": "Only a title", "blocks": [{"id": "not text"}] }
|
||||||
|
});
|
||||||
|
assert_eq!(
|
||||||
|
extract_tweet_text(&empty_article).as_deref(),
|
||||||
|
Some("https://t.co/fallback")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn build_summary_input_joins_article_backed_tweet_artifacts() {
|
||||||
|
let temp = tempfile::tempdir().unwrap();
|
||||||
|
let paths = crate::archive::initialize_archive(
|
||||||
|
temp.path(),
|
||||||
|
&temp.path().join("store"),
|
||||||
|
"Test archive",
|
||||||
|
false,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||||
|
let user_id = database::ensure_default_user(&conn).unwrap();
|
||||||
|
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
|
||||||
|
let source_id = database::upsert_source_identity(
|
||||||
|
&conn,
|
||||||
|
"twitter",
|
||||||
|
"tweet_thread",
|
||||||
|
Some("thread-1"),
|
||||||
|
Some("https://x.com/example/status/1"),
|
||||||
|
"x:thread-1",
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let entry = database::create_archived_entry(
|
||||||
|
&conn,
|
||||||
|
&database::NewEntry {
|
||||||
|
source_identity_id: source_id,
|
||||||
|
archive_run_id: run.id,
|
||||||
|
parent_entry_id: None,
|
||||||
|
root_entry_id: None,
|
||||||
|
created_by_user_id: user_id,
|
||||||
|
owned_by_user_id: user_id,
|
||||||
|
source_kind: "twitter".to_string(),
|
||||||
|
entity_kind: "tweet_thread".to_string(),
|
||||||
|
title: Some("Article thread".to_string()),
|
||||||
|
visibility: "private".to_string(),
|
||||||
|
representation_kind: "tweet_thread".to_string(),
|
||||||
|
source_metadata_json: "{}".to_string(),
|
||||||
|
display_metadata_json: None,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
for (ordinal, (relpath, body)) in [
|
||||||
|
("raw_tweets/article-one.json", "First article body."),
|
||||||
|
("raw_tweets/article-two.json", "Second article body."),
|
||||||
|
]
|
||||||
|
.into_iter()
|
||||||
|
.enumerate()
|
||||||
|
{
|
||||||
|
let path = paths.store_path.join(relpath);
|
||||||
|
std::fs::write(
|
||||||
|
&path,
|
||||||
|
serde_json::json!({
|
||||||
|
"full_text": format!("https://t.co/{ordinal}"),
|
||||||
|
"article": { "title": format!("Article {}", ordinal + 1), "plain_text": body }
|
||||||
|
})
|
||||||
|
.to_string(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
database::add_entry_artifact(
|
||||||
|
&conn,
|
||||||
|
&database::NewArtifact {
|
||||||
|
entry_id: entry.id,
|
||||||
|
artifact_role: "raw_tweet_json".to_string(),
|
||||||
|
storage_area: "raw_tweets".to_string(),
|
||||||
|
relpath: relpath.to_string(),
|
||||||
|
blob_id: None,
|
||||||
|
logical_path: None,
|
||||||
|
metadata_json: None,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
let input = build_summary_input(&paths, &entry.entry_uid).unwrap();
|
||||||
|
assert!(input.request.content.contains("First article body.\n\n---\n\nArticle 2\n\nSecond article body."));
|
||||||
|
}
|
||||||
|
|
||||||
// ── Output normalization ───────────────────────────────────────────────
|
// ── Output normalization ───────────────────────────────────────────────
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue