1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-07-22 03:05:32 +02:00

feat: expand t.co links; linkify bare URLs in tweet and article text

Frontend:
- resolveEntityBounds: try multiple candidate strings in order (u.url
  first, since that's the t.co short URL that appears in full_text)
- normalizeUrlAnn: multi-candidate search; href = expanded > url,
  display = display_url > expanded > url
- linkifyText(): regex linkifier for entity-less bare URLs; trims
  trailing punctuation [.,;:!?)] before linking; used in both
  renderTweetTextJSX and renderInlineJSX including their early-return
  paths (anns.length === 0) that previously bypassed linkification
- renderInlineJSX: fix mention href mention.name → screen_name;
  replace t.co segment text with url.display when entity covers it

Scraper (vendor/twitter/scrape_user_tweet_contents.py):
- extract_tweet_data: when note_tweet text is used, pull urls/mentions/
  hashtags/symbols from note_result.entity_set (correct indices for the
  note text); keep media from legacy.entities (no note media downloads)
This commit is contained in:
TheGeneralist 2026-07-12 15:34:39 +02:00
parent 900e33fa60
commit a06c541605
Signed by: thegeneralist01
SSH key fingerprint: SHA256:pp9qddbCNmVNoSjevdvQvM5z0DHN7LTa8qBMbcMq/R4
4 changed files with 113 additions and 50 deletions

View file

@ -4,7 +4,7 @@
<meta charset="utf-8" /> <meta charset="utf-8" />
<meta name="viewport" content="width=device-width, initial-scale=1" /> <meta name="viewport" content="width=device-width, initial-scale=1" />
<title>Archivr</title> <title>Archivr</title>
<script type="module" crossorigin src="/assets/index-C98W3QeO.js"></script> <script type="module" crossorigin src="/assets/index-WdvYwWGH.js"></script>
<link rel="stylesheet" crossorigin href="/assets/index-C89vW3ol.css"> <link rel="stylesheet" crossorigin href="/assets/index-C89vW3ol.css">
</head> </head>
<body> <body>

View file

@ -458,33 +458,80 @@ function XLogo() {
// links and @mentions with x.com links. Returns an array of React nodes. // links and @mentions with x.com links. Returns an array of React nodes.
// white-space: pre-line on the container preserves newlines in plain segments. // white-space: pre-line on the container preserves newlines in plain segments.
// Resolve start/end offsets for a URL or mention entity, handling three
// storage formats: camelCase fromIndex/toIndex (article inline), Twitter
// native indices:[s,e] array, and fallback exact-string search in fullText.
function resolveEntityBounds(ent, fullText, ...candidates) {
if (ent.fromIndex != null && ent.toIndex != null)
return { s: ent.fromIndex, e: ent.toIndex };
if (ent.indices?.length === 2)
return { s: ent.indices[0], e: ent.indices[1] };
if (fullText) {
for (const c of candidates) {
if (!c) continue;
const idx = fullText.indexOf(c);
if (idx !== -1) return { s: idx, e: idx + c.length };
}
}
return null;
}
// Build a canonical URL annotation from any entity format.
// Search candidates: short url first (appears in fullText), then expanded.
function normalizeUrlAnn(u, fullText) {
const href = u.expanded_url || u.url || u.text || '';
const display = u.display_url || u.expanded_url || u.url || u.text || '';
const bounds = resolveEntityBounds(u, fullText, u.url, u.text, u.display_url, u.expanded_url);
if (!bounds || !href) return null;
return { ...bounds, kind: 'url', href, display };
}
// Build a canonical mention annotation from any entity format.
function normalizeMentionAnn(m, fullText) {
const screen_name = m.screen_name || m.name || m.text || '';
const matchStr = screen_name ? `@${screen_name}` : null;
const bounds = resolveEntityBounds(m, fullText, matchStr);
if (!bounds || !screen_name) return null;
return { ...bounds, kind: 'mention', screen_name };
}
// Linkify bare http(s) URLs in a plain-text string that have no entity coverage.
// Returns an array of strings and <a> nodes, or the original string if no URLs.
const URL_RE = /https?:\/\/[^\s<>"'\]]+/g;
// Characters that commonly trail a URL but aren't part of it
const TRAIL_PUNCT = /[.,;:!?)]+$/;
function linkifyText(text, linkStyle) {
const parts = [];
let last = 0;
let m;
URL_RE.lastIndex = 0;
while ((m = URL_RE.exec(text)) !== null) {
if (m.index > last) parts.push(text.slice(last, m.index));
let href = m[0].replace(TRAIL_PUNCT, '');
parts.push(
<a key={m.index} href={href} target="_blank" rel="noopener noreferrer" style={linkStyle}>
{href}
</a>
);
// Put back any trimmed trailing chars as plain text
const trail = m[0].slice(href.length);
if (trail) parts.push(trail);
last = m.index + m[0].length;
}
if (last === 0) return text; // no URLs return string directly (no array alloc)
if (last < text.length) parts.push(text.slice(last));
return parts;
}
function renderTweetTextJSX(fullText, entities) { function renderTweetTextJSX(fullText, entities) {
if (!fullText) return null; if (!fullText) return null;
const urls = (entities.urls || []).filter(
u => u.fromIndex != null && u.toIndex != null
);
const mentions = (entities.user_mentions || []).filter(
m => m.fromIndex != null && m.toIndex != null
);
if (urls.length === 0 && mentions.length === 0) return decodeEnt(fullText);
const anns = [ const anns = [
...urls.map(u => ({ ...(entities.urls || []).map(u => normalizeUrlAnn(u, fullText)).filter(Boolean),
s: u.fromIndex, ...(entities.user_mentions || []).map(m => normalizeMentionAnn(m, fullText)).filter(Boolean),
e: u.toIndex,
kind: 'url',
href: u.expanded_url,
display: u.display_url,
})),
...mentions.map(m => ({
s: m.fromIndex,
e: m.toIndex,
kind: 'mention',
screen_name: m.screen_name,
})),
]; ];
if (anns.length === 0) return linkifyText(decodeEnt(fullText), S.link);
const pts = new Set([0, fullText.length]); const pts = new Set([0, fullText.length]);
for (const a of anns) { for (const a of anns) {
@ -522,7 +569,7 @@ function renderTweetTextJSX(fullText, entities) {
); );
} }
return <span key={i}>{decodeEnt(seg)}</span>; return <span key={i}>{linkifyText(decodeEnt(seg), S.link)}</span>;
}); });
} }
@ -542,12 +589,16 @@ function renderInlineJSX(text, styleRanges, urls, mentions) {
if (r.length > 0) if (r.length > 0)
anns.push({ s: r.offset, e: r.offset + r.length, kind: 'style', style: r.style }); anns.push({ s: r.offset, e: r.offset + r.length, kind: 'style', style: r.style });
} }
for (const u of urls) for (const u of urls) {
anns.push({ s: u.fromIndex, e: u.toIndex, kind: 'url', href: u.text }); const ann = normalizeUrlAnn(u, text);
for (const m of mentions) if (ann) anns.push(ann);
anns.push({ s: m.fromIndex, e: m.toIndex, kind: 'mention', name: m.text }); }
for (const m of mentions) {
const ann = normalizeMentionAnn(m, text);
if (ann) anns.push(ann);
}
if (anns.length === 0) return text; if (anns.length === 0) return linkifyText(text, S.link);
const pts = new Set([0, text.length]); const pts = new Set([0, text.length]);
for (const a of anns) { for (const a of anns) {
@ -587,9 +638,12 @@ function renderInlineJSX(text, styleRanges, urls, mentions) {
// Links wrap outermost. // Links wrap outermost.
const url = active.find(a => a.kind === 'url'); const url = active.find(a => a.kind === 'url');
if (url) { if (url) {
// If the visible segment is a raw t.co short URL, replace it with the
// human-readable display URL; otherwise keep the styled anchor text.
const isTco = /^https?:\/\/t\.co\//i.test(content);
content = ( content = (
<a href={url.href} target="_blank" rel="noopener noreferrer" style={S.link}> <a href={url.href} target="_blank" rel="noopener noreferrer" style={S.link}>
{content} {isTco ? url.display : content}
</a> </a>
); );
} }
@ -598,7 +652,7 @@ function renderInlineJSX(text, styleRanges, urls, mentions) {
if (mention) { if (mention) {
content = ( content = (
<a <a
href={`https://x.com/${mention.name}`} href={`https://x.com/${mention.screen_name}`}
target="_blank" target="_blank"
rel="noopener noreferrer" rel="noopener noreferrer"
style={S.link} style={S.link}
@ -608,7 +662,7 @@ function renderInlineJSX(text, styleRanges, urls, mentions) {
); );
} }
return <span key={i}>{content}</span>; return <span key={i}>{typeof content === 'string' ? linkifyText(content, S.link) : content}</span>;
}); });
} }

View file

@ -493,14 +493,23 @@ def extract_tweet_data(
# Extract is_quote_status (bare) # Extract is_quote_status (bare)
tweet_data["is_quote_status"] = legacy.get("is_quote_status", False) tweet_data["is_quote_status"] = legacy.get("is_quote_status", False)
# Extract entities (always included) # Extract entities - when note_tweet text is used, its entity_set has the
entities = legacy.get("entities", {}) # correct indices for that text; legacy.entities indices match legacy.full_text
# (truncated) and will be wrong/missing for long tweets.
note_result = (
tweet_result.get("note_tweet", {})
.get("note_tweet_results", {})
.get("result", {})
)
note_ents = note_result.get("entity_set", {}) if note_tweet_text else {}
legacy_ents = legacy.get("entities", {})
tweet_data["entities"] = { tweet_data["entities"] = {
"hashtags": entities.get("hashtags", []), "hashtags": note_ents.get("hashtags", legacy_ents.get("hashtags", [])),
"urls": entities.get("urls", []), "urls": note_ents.get("urls", legacy_ents.get("urls", [])),
"user_mentions": entities.get("user_mentions", []), "user_mentions": note_ents.get("user_mentions", legacy_ents.get("user_mentions", [])),
"symbols": entities.get("symbols", []), "symbols": note_ents.get("symbols", legacy_ents.get("symbols", [])),
"media": entities.get("media", []) if not bare_scrape else [], # media always from legacy (note_tweet has no media downloads)
"media": legacy_ents.get("media", []) if not bare_scrape else [],
} }
# Extract optional fields if not bare scrape # Extract optional fields if not bare scrape