mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 21:03:17 +02:00
fix(core): extract HTML title after font stripping, not before
SingleFile embeds fonts as base64 data URIs in <style> blocks in the <head>, pushing the <title> tag to ~1.2 MB in the raw temp file. The 256 KiB read window in extract_html_title missed it. Font extraction rewrites the HTML in-place (2.4 MB → 1.26 MB) before hashing, so the title is accessible at byte ~106 KB in that content. Fix: add extract_html_title_str() that operates on a &str; call it on the in-memory rewritten string after font extraction (server path). CLI path (no font extraction) falls back to result.title as before.
This commit is contained in:
parent
f9759c08a3
commit
331fe7fd61
2 changed files with 22 additions and 6 deletions
|
|
@ -691,7 +691,20 @@ fn extract_html_title(path: &Path) -> Option<String> {
|
|||
let mut f = std::fs::File::open(path).ok()?;
|
||||
let mut buf = Vec::new();
|
||||
f.take(256 * 1024).read_to_end(&mut buf).ok()?;
|
||||
let lower = String::from_utf8_lossy(&buf).to_ascii_lowercase();
|
||||
extract_html_title_from_buf(&buf)
|
||||
}
|
||||
|
||||
/// Extracts the `<title>` content from an HTML string.
|
||||
///
|
||||
/// Used in the server capture path where the font-extracted HTML is already
|
||||
/// in memory — avoids a re-read and operates on the smaller post-extraction
|
||||
/// content where the title is guaranteed to be within range.
|
||||
pub fn extract_html_title_str(html: &str) -> Option<String> {
|
||||
extract_html_title_from_buf(html.as_bytes())
|
||||
}
|
||||
|
||||
fn extract_html_title_from_buf(buf: &[u8]) -> Option<String> {
|
||||
let lower = String::from_utf8_lossy(buf).to_ascii_lowercase();
|
||||
let start = lower.find("<title>")? + "<title>".len();
|
||||
let end = lower[start..].find("</title>")? + start;
|
||||
let title = String::from_utf8_lossy(&buf[start..end])
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue