mirror of
https://github.com/thegeneralist01/archivr
synced 2026-07-22 03:05:32 +02:00
fix(core): extract HTML title after font stripping, not before
SingleFile embeds fonts as base64 data URIs in <style> blocks in the <head>, pushing the <title> tag to ~1.2 MB in the raw temp file. The 256 KiB read window in extract_html_title missed it. Font extraction rewrites the HTML in-place (2.4 MB → 1.26 MB) before hashing, so the title is accessible at byte ~106 KB in that content. Fix: add extract_html_title_str() that operates on a &str; call it on the in-memory rewritten string after font extraction (server path). CLI path (no font extraction) falls back to result.title as before.
This commit is contained in:
parent
f9759c08a3
commit
331fe7fd61
2 changed files with 22 additions and 6 deletions
|
|
@ -1065,23 +1065,26 @@ pub fn perform_capture(
|
||||||
// Font extraction: rewrite the HTML in-place before hashing.
|
// Font extraction: rewrite the HTML in-place before hashing.
|
||||||
// Only runs when archive_id is known (server context). CLI passes
|
// Only runs when archive_id is known (server context). CLI passes
|
||||||
// None and keeps fonts embedded — no behaviour change for CLI.
|
// None and keeps fonts embedded — no behaviour change for CLI.
|
||||||
let (html_hash, byte_size, extracted_fonts) =
|
let (html_hash, byte_size, extracted_fonts, html_title) =
|
||||||
if let Some(aid) = archive_id {
|
if let Some(aid) = archive_id {
|
||||||
let content = fs::read_to_string(&temp_html)
|
let content = fs::read_to_string(&temp_html)
|
||||||
.with_context(|| format!("failed to read {}", temp_html.display()))?;
|
.with_context(|| format!("failed to read {}", temp_html.display()))?;
|
||||||
let (rewritten, fonts) =
|
let (rewritten, fonts) =
|
||||||
downloader::font_extractor::extract_and_rewrite(&content, store_path, aid)
|
downloader::font_extractor::extract_and_rewrite(&content, store_path, aid)
|
||||||
.unwrap_or_else(|_| (content.clone(), vec![])); // non-fatal
|
.unwrap_or_else(|_| (content.clone(), vec![])); // non-fatal
|
||||||
|
// Extract title after font-stripping so the title tag is not buried
|
||||||
|
// behind multi-MB embedded font data that would exceed the 256 KiB window.
|
||||||
|
let title = downloader::singlefile::extract_html_title_str(&rewritten);
|
||||||
fs::write(&temp_html, rewritten.as_bytes())
|
fs::write(&temp_html, rewritten.as_bytes())
|
||||||
.with_context(|| "failed to write rewritten HTML")?;
|
.with_context(|| "failed to write rewritten HTML")?;
|
||||||
let size = rewritten.len() as i64;
|
let size = rewritten.len() as i64;
|
||||||
let new_hash = crate::hash::hash_bytes(rewritten.as_bytes());
|
let new_hash = crate::hash::hash_bytes(rewritten.as_bytes());
|
||||||
(new_hash, size, fonts)
|
(new_hash, size, fonts, title)
|
||||||
} else {
|
} else {
|
||||||
let size = fs::metadata(&temp_html)
|
let size = fs::metadata(&temp_html)
|
||||||
.with_context(|| format!("failed to stat {}", temp_html.display()))?
|
.with_context(|| format!("failed to stat {}", temp_html.display()))?
|
||||||
.len() as i64;
|
.len() as i64;
|
||||||
(result.html_hash.clone(), size, vec![])
|
(result.html_hash.clone(), size, vec![], result.title.clone())
|
||||||
};
|
};
|
||||||
|
|
||||||
// 1. Move HTML to raw store (if this hash hasn't been seen before).
|
// 1. Move HTML to raw store (if this hash hasn't been seen before).
|
||||||
|
|
@ -1120,14 +1123,14 @@ pub fn perform_capture(
|
||||||
// 4. Create the entry + primary_media artifact.
|
// 4. Create the entry + primary_media artifact.
|
||||||
// Strip mirror branding from title when fetched via Freedium.
|
// Strip mirror branding from title when fetched via Freedium.
|
||||||
let entry_title = if is_freedium_fetch {
|
let entry_title = if is_freedium_fetch {
|
||||||
result.title.as_deref().map(|t| {
|
html_title.as_deref().map(|t| {
|
||||||
t.trim_end_matches(" - Freedium")
|
t.trim_end_matches(" - Freedium")
|
||||||
.trim_end_matches(" \u{2014} Freedium")
|
.trim_end_matches(" \u{2014} Freedium")
|
||||||
.trim()
|
.trim()
|
||||||
.to_string()
|
.to_string()
|
||||||
})
|
})
|
||||||
} else {
|
} else {
|
||||||
result.title
|
html_title
|
||||||
};
|
};
|
||||||
let entry = record_media_entry(
|
let entry = record_media_entry(
|
||||||
&conn,
|
&conn,
|
||||||
|
|
|
||||||
|
|
@ -691,7 +691,20 @@ fn extract_html_title(path: &Path) -> Option<String> {
|
||||||
let mut f = std::fs::File::open(path).ok()?;
|
let mut f = std::fs::File::open(path).ok()?;
|
||||||
let mut buf = Vec::new();
|
let mut buf = Vec::new();
|
||||||
f.take(256 * 1024).read_to_end(&mut buf).ok()?;
|
f.take(256 * 1024).read_to_end(&mut buf).ok()?;
|
||||||
let lower = String::from_utf8_lossy(&buf).to_ascii_lowercase();
|
extract_html_title_from_buf(&buf)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Extracts the `<title>` content from an HTML string.
|
||||||
|
///
|
||||||
|
/// Used in the server capture path where the font-extracted HTML is already
|
||||||
|
/// in memory — avoids a re-read and operates on the smaller post-extraction
|
||||||
|
/// content where the title is guaranteed to be within range.
|
||||||
|
pub fn extract_html_title_str(html: &str) -> Option<String> {
|
||||||
|
extract_html_title_from_buf(html.as_bytes())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn extract_html_title_from_buf(buf: &[u8]) -> Option<String> {
|
||||||
|
let lower = String::from_utf8_lossy(buf).to_ascii_lowercase();
|
||||||
let start = lower.find("<title>")? + "<title>".len();
|
let start = lower.find("<title>")? + "<title>".len();
|
||||||
let end = lower[start..].find("</title>")? + start;
|
let end = lower[start..].find("</title>")? + start;
|
||||||
let title = String::from_utf8_lossy(&buf[start..end])
|
let title = String::from_utf8_lossy(&buf[start..end])
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue