mirror of
https://github.com/thegeneralist01/archivr
synced 2026-07-22 03:05:32 +02:00
feat: expand t.co links; linkify bare URLs in tweet and article text
Frontend: - resolveEntityBounds: try multiple candidate strings in order (u.url first, since that's the t.co short URL that appears in full_text) - normalizeUrlAnn: multi-candidate search; href = expanded > url, display = display_url > expanded > url - linkifyText(): regex linkifier for entity-less bare URLs; trims trailing punctuation [.,;:!?)] before linking; used in both renderTweetTextJSX and renderInlineJSX including their early-return paths (anns.length === 0) that previously bypassed linkification - renderInlineJSX: fix mention href mention.name → screen_name; replace t.co segment text with url.display when entity covers it Scraper (vendor/twitter/scrape_user_tweet_contents.py): - extract_tweet_data: when note_tweet text is used, pull urls/mentions/ hashtags/symbols from note_result.entity_set (correct indices for the note text); keep media from legacy.entities (no note media downloads)
This commit is contained in:
parent
900e33fa60
commit
a06c541605
4 changed files with 113 additions and 50 deletions
File diff suppressed because one or more lines are too long
|
|
@ -4,7 +4,7 @@
|
||||||
<meta charset="utf-8" />
|
<meta charset="utf-8" />
|
||||||
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
||||||
<title>Archivr</title>
|
<title>Archivr</title>
|
||||||
<script type="module" crossorigin src="/assets/index-C98W3QeO.js"></script>
|
<script type="module" crossorigin src="/assets/index-WdvYwWGH.js"></script>
|
||||||
<link rel="stylesheet" crossorigin href="/assets/index-C89vW3ol.css">
|
<link rel="stylesheet" crossorigin href="/assets/index-C89vW3ol.css">
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
|
|
|
||||||
|
|
@ -458,33 +458,80 @@ function XLogo() {
|
||||||
// links and @mentions with x.com links. Returns an array of React nodes.
|
// links and @mentions with x.com links. Returns an array of React nodes.
|
||||||
// white-space: pre-line on the container preserves newlines in plain segments.
|
// white-space: pre-line on the container preserves newlines in plain segments.
|
||||||
|
|
||||||
|
// Resolve start/end offsets for a URL or mention entity, handling three
|
||||||
|
// storage formats: camelCase fromIndex/toIndex (article inline), Twitter
|
||||||
|
// native indices:[s,e] array, and fallback exact-string search in fullText.
|
||||||
|
function resolveEntityBounds(ent, fullText, ...candidates) {
|
||||||
|
if (ent.fromIndex != null && ent.toIndex != null)
|
||||||
|
return { s: ent.fromIndex, e: ent.toIndex };
|
||||||
|
if (ent.indices?.length === 2)
|
||||||
|
return { s: ent.indices[0], e: ent.indices[1] };
|
||||||
|
if (fullText) {
|
||||||
|
for (const c of candidates) {
|
||||||
|
if (!c) continue;
|
||||||
|
const idx = fullText.indexOf(c);
|
||||||
|
if (idx !== -1) return { s: idx, e: idx + c.length };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Build a canonical URL annotation from any entity format.
|
||||||
|
// Search candidates: short url first (appears in fullText), then expanded.
|
||||||
|
function normalizeUrlAnn(u, fullText) {
|
||||||
|
const href = u.expanded_url || u.url || u.text || '';
|
||||||
|
const display = u.display_url || u.expanded_url || u.url || u.text || '';
|
||||||
|
const bounds = resolveEntityBounds(u, fullText, u.url, u.text, u.display_url, u.expanded_url);
|
||||||
|
if (!bounds || !href) return null;
|
||||||
|
return { ...bounds, kind: 'url', href, display };
|
||||||
|
}
|
||||||
|
|
||||||
|
// Build a canonical mention annotation from any entity format.
|
||||||
|
function normalizeMentionAnn(m, fullText) {
|
||||||
|
const screen_name = m.screen_name || m.name || m.text || '';
|
||||||
|
const matchStr = screen_name ? `@${screen_name}` : null;
|
||||||
|
const bounds = resolveEntityBounds(m, fullText, matchStr);
|
||||||
|
if (!bounds || !screen_name) return null;
|
||||||
|
return { ...bounds, kind: 'mention', screen_name };
|
||||||
|
}
|
||||||
|
|
||||||
|
// Linkify bare http(s) URLs in a plain-text string that have no entity coverage.
|
||||||
|
// Returns an array of strings and <a> nodes, or the original string if no URLs.
|
||||||
|
const URL_RE = /https?:\/\/[^\s<>"'\])]+/g;
|
||||||
|
// Characters that commonly trail a URL but aren't part of it
|
||||||
|
const TRAIL_PUNCT = /[.,;:!?)()]+$/;
|
||||||
|
function linkifyText(text, linkStyle) {
|
||||||
|
const parts = [];
|
||||||
|
let last = 0;
|
||||||
|
let m;
|
||||||
|
URL_RE.lastIndex = 0;
|
||||||
|
while ((m = URL_RE.exec(text)) !== null) {
|
||||||
|
if (m.index > last) parts.push(text.slice(last, m.index));
|
||||||
|
let href = m[0].replace(TRAIL_PUNCT, '');
|
||||||
|
parts.push(
|
||||||
|
<a key={m.index} href={href} target="_blank" rel="noopener noreferrer" style={linkStyle}>
|
||||||
|
{href}
|
||||||
|
</a>
|
||||||
|
);
|
||||||
|
// Put back any trimmed trailing chars as plain text
|
||||||
|
const trail = m[0].slice(href.length);
|
||||||
|
if (trail) parts.push(trail);
|
||||||
|
last = m.index + m[0].length;
|
||||||
|
}
|
||||||
|
if (last === 0) return text; // no URLs — return string directly (no array alloc)
|
||||||
|
if (last < text.length) parts.push(text.slice(last));
|
||||||
|
return parts;
|
||||||
|
}
|
||||||
|
|
||||||
function renderTweetTextJSX(fullText, entities) {
|
function renderTweetTextJSX(fullText, entities) {
|
||||||
if (!fullText) return null;
|
if (!fullText) return null;
|
||||||
|
|
||||||
const urls = (entities.urls || []).filter(
|
|
||||||
u => u.fromIndex != null && u.toIndex != null
|
|
||||||
);
|
|
||||||
const mentions = (entities.user_mentions || []).filter(
|
|
||||||
m => m.fromIndex != null && m.toIndex != null
|
|
||||||
);
|
|
||||||
|
|
||||||
if (urls.length === 0 && mentions.length === 0) return decodeEnt(fullText);
|
|
||||||
|
|
||||||
const anns = [
|
const anns = [
|
||||||
...urls.map(u => ({
|
...(entities.urls || []).map(u => normalizeUrlAnn(u, fullText)).filter(Boolean),
|
||||||
s: u.fromIndex,
|
...(entities.user_mentions || []).map(m => normalizeMentionAnn(m, fullText)).filter(Boolean),
|
||||||
e: u.toIndex,
|
|
||||||
kind: 'url',
|
|
||||||
href: u.expanded_url,
|
|
||||||
display: u.display_url,
|
|
||||||
})),
|
|
||||||
...mentions.map(m => ({
|
|
||||||
s: m.fromIndex,
|
|
||||||
e: m.toIndex,
|
|
||||||
kind: 'mention',
|
|
||||||
screen_name: m.screen_name,
|
|
||||||
})),
|
|
||||||
];
|
];
|
||||||
|
if (anns.length === 0) return linkifyText(decodeEnt(fullText), S.link);
|
||||||
|
|
||||||
|
|
||||||
const pts = new Set([0, fullText.length]);
|
const pts = new Set([0, fullText.length]);
|
||||||
for (const a of anns) {
|
for (const a of anns) {
|
||||||
|
|
@ -522,7 +569,7 @@ function renderTweetTextJSX(fullText, entities) {
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
return <span key={i}>{decodeEnt(seg)}</span>;
|
return <span key={i}>{linkifyText(decodeEnt(seg), S.link)}</span>;
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -542,12 +589,16 @@ function renderInlineJSX(text, styleRanges, urls, mentions) {
|
||||||
if (r.length > 0)
|
if (r.length > 0)
|
||||||
anns.push({ s: r.offset, e: r.offset + r.length, kind: 'style', style: r.style });
|
anns.push({ s: r.offset, e: r.offset + r.length, kind: 'style', style: r.style });
|
||||||
}
|
}
|
||||||
for (const u of urls)
|
for (const u of urls) {
|
||||||
anns.push({ s: u.fromIndex, e: u.toIndex, kind: 'url', href: u.text });
|
const ann = normalizeUrlAnn(u, text);
|
||||||
for (const m of mentions)
|
if (ann) anns.push(ann);
|
||||||
anns.push({ s: m.fromIndex, e: m.toIndex, kind: 'mention', name: m.text });
|
}
|
||||||
|
for (const m of mentions) {
|
||||||
|
const ann = normalizeMentionAnn(m, text);
|
||||||
|
if (ann) anns.push(ann);
|
||||||
|
}
|
||||||
|
|
||||||
if (anns.length === 0) return text;
|
if (anns.length === 0) return linkifyText(text, S.link);
|
||||||
|
|
||||||
const pts = new Set([0, text.length]);
|
const pts = new Set([0, text.length]);
|
||||||
for (const a of anns) {
|
for (const a of anns) {
|
||||||
|
|
@ -587,9 +638,12 @@ function renderInlineJSX(text, styleRanges, urls, mentions) {
|
||||||
// Links wrap outermost.
|
// Links wrap outermost.
|
||||||
const url = active.find(a => a.kind === 'url');
|
const url = active.find(a => a.kind === 'url');
|
||||||
if (url) {
|
if (url) {
|
||||||
|
// If the visible segment is a raw t.co short URL, replace it with the
|
||||||
|
// human-readable display URL; otherwise keep the styled anchor text.
|
||||||
|
const isTco = /^https?:\/\/t\.co\//i.test(content);
|
||||||
content = (
|
content = (
|
||||||
<a href={url.href} target="_blank" rel="noopener noreferrer" style={S.link}>
|
<a href={url.href} target="_blank" rel="noopener noreferrer" style={S.link}>
|
||||||
{content}
|
{isTco ? url.display : content}
|
||||||
</a>
|
</a>
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
@ -598,7 +652,7 @@ function renderInlineJSX(text, styleRanges, urls, mentions) {
|
||||||
if (mention) {
|
if (mention) {
|
||||||
content = (
|
content = (
|
||||||
<a
|
<a
|
||||||
href={`https://x.com/${mention.name}`}
|
href={`https://x.com/${mention.screen_name}`}
|
||||||
target="_blank"
|
target="_blank"
|
||||||
rel="noopener noreferrer"
|
rel="noopener noreferrer"
|
||||||
style={S.link}
|
style={S.link}
|
||||||
|
|
@ -608,7 +662,7 @@ function renderInlineJSX(text, styleRanges, urls, mentions) {
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
return <span key={i}>{content}</span>;
|
return <span key={i}>{typeof content === 'string' ? linkifyText(content, S.link) : content}</span>;
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
23
vendor/twitter/scrape_user_tweet_contents.py
vendored
23
vendor/twitter/scrape_user_tweet_contents.py
vendored
|
|
@ -493,14 +493,23 @@ def extract_tweet_data(
|
||||||
# Extract is_quote_status (bare)
|
# Extract is_quote_status (bare)
|
||||||
tweet_data["is_quote_status"] = legacy.get("is_quote_status", False)
|
tweet_data["is_quote_status"] = legacy.get("is_quote_status", False)
|
||||||
|
|
||||||
# Extract entities (always included)
|
# Extract entities - when note_tweet text is used, its entity_set has the
|
||||||
entities = legacy.get("entities", {})
|
# correct indices for that text; legacy.entities indices match legacy.full_text
|
||||||
|
# (truncated) and will be wrong/missing for long tweets.
|
||||||
|
note_result = (
|
||||||
|
tweet_result.get("note_tweet", {})
|
||||||
|
.get("note_tweet_results", {})
|
||||||
|
.get("result", {})
|
||||||
|
)
|
||||||
|
note_ents = note_result.get("entity_set", {}) if note_tweet_text else {}
|
||||||
|
legacy_ents = legacy.get("entities", {})
|
||||||
tweet_data["entities"] = {
|
tweet_data["entities"] = {
|
||||||
"hashtags": entities.get("hashtags", []),
|
"hashtags": note_ents.get("hashtags", legacy_ents.get("hashtags", [])),
|
||||||
"urls": entities.get("urls", []),
|
"urls": note_ents.get("urls", legacy_ents.get("urls", [])),
|
||||||
"user_mentions": entities.get("user_mentions", []),
|
"user_mentions": note_ents.get("user_mentions", legacy_ents.get("user_mentions", [])),
|
||||||
"symbols": entities.get("symbols", []),
|
"symbols": note_ents.get("symbols", legacy_ents.get("symbols", [])),
|
||||||
"media": entities.get("media", []) if not bare_scrape else [],
|
# media always from legacy (note_tweet has no media downloads)
|
||||||
|
"media": legacy_ents.get("media", []) if not bare_scrape else [],
|
||||||
}
|
}
|
||||||
|
|
||||||
# Extract optional fields if not bare scrape
|
# Extract optional fields if not bare scrape
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue