1
Fork 0
mirror of https://github.com/thegeneralist01/archivr synced 2026-10-09 21:03:17 +02:00

fix(core): extract HTML title after font stripping, not before

SingleFile embeds fonts as base64 data URIs in <style> blocks in the
<head>, pushing the <title> tag to ~1.2 MB in the raw temp file.
The 256 KiB read window in extract_html_title missed it.

Font extraction rewrites the HTML in-place (2.4 MB → 1.26 MB) before
hashing, so the title is accessible at byte ~106 KB in that content.

Fix: add extract_html_title_str() that operates on a &str; call it on
the in-memory rewritten string after font extraction (server path).
CLI path (no font extraction) falls back to result.title as before.
This commit is contained in:
TheGeneralist 2026-07-19 13:15:35 +02:00
parent f9759c08a3
commit 331fe7fd61
Signed by: thegeneralist01
SSH key fingerprint: SHA256:pp9qddbCNmVNoSjevdvQvM5z0DHN7LTa8qBMbcMq/R4
2 changed files with 22 additions and 6 deletions

View file

@ -691,7 +691,20 @@ fn extract_html_title(path: &Path) -> Option<String> {
let mut f = std::fs::File::open(path).ok()?;
let mut buf = Vec::new();
f.take(256 * 1024).read_to_end(&mut buf).ok()?;
let lower = String::from_utf8_lossy(&buf).to_ascii_lowercase();
extract_html_title_from_buf(&buf)
}
/// Extracts the `<title>` content from an HTML string.
///
/// Used in the server capture path where the font-extracted HTML is already
/// in memory — avoids a re-read and operates on the smaller post-extraction
/// content where the title is guaranteed to be within range.
pub fn extract_html_title_str(html: &str) -> Option<String> {
extract_html_title_from_buf(html.as_bytes())
}
fn extract_html_title_from_buf(buf: &[u8]) -> Option<String> {
let lower = String::from_utf8_lossy(buf).to_ascii_lowercase();
let start = lower.find("<title>")? + "<title>".len();
let end = lower[start..].find("</title>")? + start;
let title = String::from_utf8_lossy(&buf[start..end])