mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 12:55:00 +02:00
fix(core): summarize tweets + walk all tweets in a thread
Tweet and tweet_thread entries store their payload under artifact_role `raw_tweet_json`, not `primary_media`. `build_summary_input` filtered strictly for `primary_media LIMIT 1`, so both cases silently failed with 'entry X has no primary_media artifact to summarize'. Threads compound the problem: the tweet scraper writes ONE json file per status, so even a fixed lookup that took the first row would summarize only the initial tweet and lose the rest of the conversation. Fixes: - New `load_summary_artifacts` helper returns every artifact for a role in insertion order. - For entity_kind `tweet` / `tweet_thread`, load all `raw_tweet_json` artifacts (falling back to `primary_media` for archives predating that role convention). - Iterate artifacts, extract text per file with the existing markdown/html/json branches, then join thread pieces with a `---` separator so the model sees a real paragraph break between statuses instead of one flowing document. Single-tweet entries produce one piece and the separator never renders. Non-tweet entries behave exactly as before.
This commit is contained in:
parent
a4fb2ed795
commit
189fe2d392
1 changed files with 68 additions and 37 deletions
|
|
@ -736,6 +736,26 @@ fn extension_of(relpath: &str) -> String {
|
||||||
/// Returns a descriptive error rather than a summary for artifact kinds v1 does
|
/// Returns a descriptive error rather than a summary for artifact kinds v1 does
|
||||||
/// not handle (video, audio, images): the caller surfaces it to the UI, and a
|
/// not handle (video, audio, images): the caller surfaces it to the UI, and a
|
||||||
/// clear "unsupported" beats an empty or hallucinated summary.
|
/// clear "unsupported" beats an empty or hallucinated summary.
|
||||||
|
fn load_summary_artifacts(
|
||||||
|
conn: &rusqlite::Connection,
|
||||||
|
entry_id: i64,
|
||||||
|
artifact_role: &str,
|
||||||
|
) -> Result<Vec<(String, Option<String>)>> {
|
||||||
|
let mut stmt = conn.prepare(
|
||||||
|
"SELECT ea.relpath, b.mime_type
|
||||||
|
FROM entry_artifacts ea
|
||||||
|
LEFT JOIN blobs b ON b.id = ea.blob_id
|
||||||
|
WHERE ea.entry_id = ?1 AND ea.artifact_role = ?2
|
||||||
|
ORDER BY ea.id ASC",
|
||||||
|
)?;
|
||||||
|
let rows = stmt
|
||||||
|
.query_map(rusqlite::params![entry_id, artifact_role], |row| {
|
||||||
|
Ok((row.get::<_, String>(0)?, row.get::<_, Option<String>>(1)?))
|
||||||
|
})?
|
||||||
|
.collect::<rusqlite::Result<Vec<_>>>()?;
|
||||||
|
Ok(rows)
|
||||||
|
}
|
||||||
|
|
||||||
pub fn build_summary_input(paths: &ArchivePaths, entry_uid: &str) -> Result<SummaryInput> {
|
pub fn build_summary_input(paths: &ArchivePaths, entry_uid: &str) -> Result<SummaryInput> {
|
||||||
let conn = database::open_or_initialize(&paths.archive_path)?;
|
let conn = database::open_or_initialize(&paths.archive_path)?;
|
||||||
let (entry_id, title, source_kind, entity_kind) = conn
|
let (entry_id, title, source_kind, entity_kind) = conn
|
||||||
|
|
@ -754,23 +774,30 @@ pub fn build_summary_input(paths: &ArchivePaths, entry_uid: &str) -> Result<Summ
|
||||||
)
|
)
|
||||||
.map_err(|_| anyhow!("entry not found: {entry_uid}"))?;
|
.map_err(|_| anyhow!("entry not found: {entry_uid}"))?;
|
||||||
|
|
||||||
let (relpath, mime): (String, Option<String>) = conn
|
// Tweets and tweet threads use `raw_tweet_json` rather than `primary_media`,
|
||||||
.query_row(
|
// and a THREAD is materialized as N separate JSON files (one per status).
|
||||||
"SELECT ea.relpath, b.mime_type
|
// Load every matching artifact in insertion order so a thread summarizes
|
||||||
FROM entry_artifacts ea
|
// as the whole conversation, not just its first status.
|
||||||
LEFT JOIN blobs b ON b.id = ea.blob_id
|
let is_tweetish = matches!(entity_kind.as_str(), "tweet" | "tweet_thread");
|
||||||
WHERE ea.entry_id = ?1 AND ea.artifact_role = 'primary_media'
|
let primary_role = if is_tweetish { "raw_tweet_json" } else { "primary_media" };
|
||||||
ORDER BY ea.id ASC LIMIT 1",
|
let mut artifacts = load_summary_artifacts(&conn, entry_id, primary_role)?;
|
||||||
[entry_id],
|
if artifacts.is_empty() && is_tweetish {
|
||||||
|row| Ok((row.get(0)?, row.get(1)?)),
|
// Older archives may have stored tweet payloads under `primary_media`.
|
||||||
)
|
artifacts = load_summary_artifacts(&conn, entry_id, "primary_media")?;
|
||||||
.map_err(|_| anyhow!("entry {entry_uid} has no primary_media artifact to summarize"))?;
|
}
|
||||||
|
if artifacts.is_empty() {
|
||||||
|
bail!("entry {entry_uid} has no {primary_role} artifact to summarize");
|
||||||
|
}
|
||||||
|
|
||||||
let abs = paths.store_path.join(&relpath);
|
let mut pieces: Vec<String> = Vec::with_capacity(artifacts.len());
|
||||||
let ext = extension_of(&relpath);
|
for (relpath, mime_opt) in &artifacts {
|
||||||
let mime = mime.unwrap_or_default();
|
let abs = paths.store_path.join(relpath);
|
||||||
|
let ext = extension_of(relpath);
|
||||||
|
let mime = mime_opt.clone().unwrap_or_default();
|
||||||
|
|
||||||
let content = if ext == "md" || ext == "markdown" || ext == "txt" || mime.starts_with("text/markdown") || mime == "text/plain" {
|
let piece = if ext == "md" || ext == "markdown" || ext == "txt"
|
||||||
|
|| mime.starts_with("text/markdown") || mime == "text/plain"
|
||||||
|
{
|
||||||
std::fs::read_to_string(&abs)
|
std::fs::read_to_string(&abs)
|
||||||
.with_context(|| format!("failed to read {}", abs.display()))?
|
.with_context(|| format!("failed to read {}", abs.display()))?
|
||||||
} else if ext == "html" || ext == "htm" || mime.starts_with("text/html") {
|
} else if ext == "html" || ext == "htm" || mime.starts_with("text/html") {
|
||||||
|
|
@ -782,17 +809,21 @@ pub fn build_summary_input(paths: &ArchivePaths, entry_uid: &str) -> Result<Summ
|
||||||
.with_context(|| format!("failed to read {}", abs.display()))?;
|
.with_context(|| format!("failed to read {}", abs.display()))?;
|
||||||
let parsed: serde_json::Value = serde_json::from_str(&raw)
|
let parsed: serde_json::Value = serde_json::from_str(&raw)
|
||||||
.with_context(|| format!("{} is not valid JSON", abs.display()))?;
|
.with_context(|| format!("{} is not valid JSON", abs.display()))?;
|
||||||
extract_tweet_text(&parsed)
|
extract_tweet_text(&parsed).unwrap_or_default()
|
||||||
.ok_or_else(|| anyhow!("no text content available for this entry kind — v1 unsupported"))?
|
|
||||||
} else {
|
} else {
|
||||||
bail!(
|
bail!(
|
||||||
"no text content available for this entry kind — v1 unsupported \
|
"no text content available for this entry kind — v1 unsupported \
|
||||||
(primary artifact {relpath}, mime {})",
|
(artifact {relpath}, mime {})",
|
||||||
if mime.is_empty() { "unknown" } else { &mime }
|
if mime.is_empty() { "unknown" } else { &mime }
|
||||||
);
|
);
|
||||||
};
|
};
|
||||||
|
if !piece.trim().is_empty() {
|
||||||
let content = content.trim().to_string();
|
pieces.push(piece);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Thread joiner: `---` on its own line reads as a paragraph break to both
|
||||||
|
// humans and models. Single-piece entries never render the separator.
|
||||||
|
let content = pieces.join("\n\n---\n\n").trim().to_string();
|
||||||
if content.is_empty() {
|
if content.is_empty() {
|
||||||
bail!("no text content available for this entry kind — v1 unsupported (extracted text was empty)");
|
bail!("no text content available for this entry kind — v1 unsupported (extracted text was empty)");
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue