mirror of
https://github.com/thegeneralist01/archivr
synced 2026-07-21 18:55:36 +02:00
test(core): extract helpers and add unit tests for Codex review fixes
Extract two testable pure functions from inline_archivr_img_srcs: - html_attr_decode: decodes HTML character references in attribute values. Fixes decode order: & runs LAST so &lt; → < (one layer removed), not < (two layers). Previous order caused double-decoding. - same_origin_cookie_header: returns a Cookie header value only when img_url's domain matches capture_url's domain. Add 11 unit tests covering: - html_attr_decode: plain URL no-op, & in CDN query params, single-layer decode (&lt; → < not <), double-encoded amp (&amp; → &), direct </>/" decode - same_origin_cookie_header: same host attaches cookies, third-party returns None, empty cookies returns None, Freedium empty-cookies no-op - bounded_read: Read::take stops at max+1 bytes (guard fires), allows exactly-at-limit payloads (guard does not fire)
This commit is contained in:
parent
8c19d90e94
commit
e5fb6ea9bb
1 changed files with 160 additions and 26 deletions
|
|
@ -841,6 +841,46 @@ fn mime_to_favicon_ext(mime: &str) -> Option<&'static str> {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Decodes the minimal set of HTML character references that browsers encode
|
||||||
|
/// in attribute values during serialisation. Covers the four characters
|
||||||
|
/// browsers must escape (`&`, `<`, `>`, `"`) — sufficient for URL query
|
||||||
|
/// strings and signed CDN parameters embedded in `data-*` attributes.
|
||||||
|
fn html_attr_decode(s: &str) -> String {
|
||||||
|
// Decode & LAST so a value like `&lt;` becomes `<` (one layer
|
||||||
|
// removed) rather than `<` (two layers). If `&` ran first it would
|
||||||
|
// produce `<` from the remainder, which the subsequent `<` pass
|
||||||
|
// would then incorrectly collapse to `<`.
|
||||||
|
s.replace("<", "<")
|
||||||
|
.replace(">", ">")
|
||||||
|
.replace(""", "\"")
|
||||||
|
.replace("&", "&")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns a `Cookie: …` header value when `img_url` is same-domain as
|
||||||
|
/// `capture_url` and `cookies` is non-empty; `None` otherwise.
|
||||||
|
/// Uses `domain_from_url` for consistency with the SingleFile cookie-file writer.
|
||||||
|
fn same_origin_cookie_header(
|
||||||
|
capture_url: &str,
|
||||||
|
img_url: &str,
|
||||||
|
cookies: &HashMap<String, String>,
|
||||||
|
) -> Option<String> {
|
||||||
|
if cookies.is_empty() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let cap = domain_from_url(capture_url);
|
||||||
|
let img = domain_from_url(img_url);
|
||||||
|
if cap.is_empty() || img.is_empty() || cap != img {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(
|
||||||
|
cookies
|
||||||
|
.iter()
|
||||||
|
.map(|(k, v)| format!("{k}={v}"))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join("; "),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
/// Scans a reader-mode HTML file for `<img data-archivr-src="…">` elements whose
|
/// Scans a reader-mode HTML file for `<img data-archivr-src="…">` elements whose
|
||||||
/// `src` is not already a large inlined image, fetches those URLs with blocking
|
/// `src` is not already a large inlined image, fetches those URLs with blocking
|
||||||
/// reqwest, and replaces `src` with `data:<mime>;base64,…`. Non-fatal: any failure
|
/// reqwest, and replaces `src` with `data:<mime>;base64,…`. Non-fatal: any failure
|
||||||
|
|
@ -869,7 +909,6 @@ fn inline_archivr_img_srcs(path: &Path, capture_url: &str, cookies: &HashMap<Str
|
||||||
Err(_) => return,
|
Err(_) => return,
|
||||||
};
|
};
|
||||||
|
|
||||||
let capture_domain = domain_from_url(capture_url);
|
|
||||||
|
|
||||||
let re_img = Regex::new(r"(?si)<img\b[^>]*>").unwrap();
|
let re_img = Regex::new(r"(?si)<img\b[^>]*>").unwrap();
|
||||||
let re_archivr = Regex::new(r#"(?i)\bdata-archivr-src="([^"]+)""#).unwrap();
|
let re_archivr = Regex::new(r#"(?i)\bdata-archivr-src="([^"]+)""#).unwrap();
|
||||||
|
|
@ -887,13 +926,9 @@ fn inline_archivr_img_srcs(path: &Path, capture_url: &str, cookies: &HashMap<Str
|
||||||
Some(u) => u,
|
Some(u) => u,
|
||||||
None => continue,
|
None => continue,
|
||||||
};
|
};
|
||||||
// HTML-decode common entities that the browser's serialiser encodes in
|
// HTML-decode entities the browser's serialiser encodes in attribute values
|
||||||
// attribute values (e.g. & → & in CDN signed/query URLs).
|
// (e.g. & → & in CDN signed/query URLs).
|
||||||
let img_url = raw_url
|
let img_url = html_attr_decode(&raw_url);
|
||||||
.replace("&", "&")
|
|
||||||
.replace("<", "<")
|
|
||||||
.replace(">", ">")
|
|
||||||
.replace(""", "\"");
|
|
||||||
|
|
||||||
// Already a large inlined image — leave it alone.
|
// Already a large inlined image — leave it alone.
|
||||||
let cur_src = re_src.captures(tag).map(|c| c[1].to_string()).unwrap_or_default();
|
let cur_src = re_src.captures(tag).map(|c| c[1].to_string()).unwrap_or_default();
|
||||||
|
|
@ -901,24 +936,7 @@ fn inline_archivr_img_srcs(path: &Path, capture_url: &str, cookies: &HashMap<Str
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Forward cookies only when the image host exactly matches the captured
|
let cookie_header = same_origin_cookie_header(capture_url, &img_url, cookies);
|
||||||
// page's domain. Avoids leaking credentials to third-party image hosts.
|
|
||||||
let img_domain = domain_from_url(&img_url);
|
|
||||||
let cookie_header = if !capture_domain.is_empty()
|
|
||||||
&& !img_domain.is_empty()
|
|
||||||
&& img_domain == capture_domain
|
|
||||||
&& !cookies.is_empty()
|
|
||||||
{
|
|
||||||
Some(
|
|
||||||
cookies
|
|
||||||
.iter()
|
|
||||||
.map(|(k, v)| format!("{k}={v}"))
|
|
||||||
.collect::<Vec<_>>()
|
|
||||||
.join("; "),
|
|
||||||
)
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
};
|
|
||||||
|
|
||||||
let data_uri = match fetch_image_as_data_uri(&client, &img_url, MAX_BYTES, cookie_header) {
|
let data_uri = match fetch_image_as_data_uri(&client, &img_url, MAX_BYTES, cookie_header) {
|
||||||
Ok(u) => u,
|
Ok(u) => u,
|
||||||
|
|
@ -1101,9 +1119,125 @@ mod tests {
|
||||||
enabled.eq_ignore_ascii_case("false") || enabled == "0";
|
enabled.eq_ignore_ascii_case("false") || enabled == "0";
|
||||||
assert!(is_disabled);
|
assert!(is_disabled);
|
||||||
|
|
||||||
|
|
||||||
let enabled = "0";
|
let enabled = "0";
|
||||||
let is_disabled =
|
let is_disabled =
|
||||||
enabled.eq_ignore_ascii_case("false") || enabled == "0";
|
enabled.eq_ignore_ascii_case("false") || enabled == "0";
|
||||||
assert!(is_disabled);
|
assert!(is_disabled);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ── html_attr_decode ─────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn html_attr_decode_plain_url_unchanged() {
|
||||||
|
assert_eq!(
|
||||||
|
html_attr_decode("https://cdn.example.com/img.jpg"),
|
||||||
|
"https://cdn.example.com/img.jpg",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn html_attr_decode_ampersand_in_query() {
|
||||||
|
// CDN signed URL with & encoded as & by the browser's HTML serialiser.
|
||||||
|
assert_eq!(
|
||||||
|
html_attr_decode("https://cdn.example.com/img.jpg?a=1&b=2&sig=abc"),
|
||||||
|
"https://cdn.example.com/img.jpg?a=1&b=2&sig=abc",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn html_attr_decode_single_layer_only() {
|
||||||
|
// &lt; → < (one browser-serialisation layer removed, not two).
|
||||||
|
// If & ran first it would produce < then < — wrong.
|
||||||
|
assert_eq!(html_attr_decode("&lt;"), "<");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn html_attr_decode_double_encoded_amp() {
|
||||||
|
// &amp; → & (the attribute value is a literal &).
|
||||||
|
assert_eq!(html_attr_decode("&amp;"), "&");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn html_attr_decode_lt_gt_quot_direct() {
|
||||||
|
// Directly encoded entities (no surrounding &) decode normally.
|
||||||
|
assert_eq!(html_attr_decode("<tag>"), "<tag>");
|
||||||
|
assert_eq!(html_attr_decode("say "hi""), "say \"hi\"");
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── same_origin_cookie_header ─────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn same_origin_cookie_header_same_host_attaches_cookies() {
|
||||||
|
let mut cookies = HashMap::new();
|
||||||
|
cookies.insert("session".to_string(), "abc".to_string());
|
||||||
|
cookies.insert("tok".to_string(), "xyz".to_string());
|
||||||
|
let h = same_origin_cookie_header(
|
||||||
|
"https://example.com/article",
|
||||||
|
"https://example.com/img/photo.jpg",
|
||||||
|
&cookies,
|
||||||
|
).expect("expected Some for same host");
|
||||||
|
assert!(h.contains("session=abc"), "missing session: {h}");
|
||||||
|
assert!(h.contains("tok=xyz"), "missing tok: {h}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn same_origin_cookie_header_third_party_returns_none() {
|
||||||
|
let mut cookies = HashMap::new();
|
||||||
|
cookies.insert("session".to_string(), "abc".to_string());
|
||||||
|
assert!(same_origin_cookie_header(
|
||||||
|
"https://example.com/article",
|
||||||
|
"https://cdn.third-party.com/img.jpg",
|
||||||
|
&cookies,
|
||||||
|
).is_none(), "should not forward cookies to third-party host");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn same_origin_cookie_header_empty_cookies_returns_none() {
|
||||||
|
assert!(same_origin_cookie_header(
|
||||||
|
"https://example.com/article",
|
||||||
|
"https://example.com/img.jpg",
|
||||||
|
&HashMap::new(),
|
||||||
|
).is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn same_origin_cookie_header_freedium_empty_cookies_noop() {
|
||||||
|
// capture.rs passes empty cookies for Freedium fetches — confirm no-op.
|
||||||
|
assert!(same_origin_cookie_header(
|
||||||
|
"https://freedium-mirror.cfd/https://example.com/",
|
||||||
|
"https://freedium-mirror.cfd/img/example.com/photo.jpg",
|
||||||
|
&HashMap::new(),
|
||||||
|
).is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── bounded read (size cap) ───────────────────────────────────────────────
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bounded_read_stops_at_limit() {
|
||||||
|
// Exercises the same Read::take logic used in fetch_image_as_data_uri.
|
||||||
|
let max: usize = 10;
|
||||||
|
let data: Vec<u8> = (0u8..100).collect();
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
std::io::Cursor::new(&data)
|
||||||
|
.take((max as u64) + 1)
|
||||||
|
.read_to_end(&mut buf)
|
||||||
|
.unwrap();
|
||||||
|
assert!(buf.len() > max, "take should read up to max+1");
|
||||||
|
assert!(buf.len() <= max + 1, "take should not exceed max+1");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bounded_read_allows_at_limit() {
|
||||||
|
let max: usize = 10;
|
||||||
|
let data: Vec<u8> = vec![0u8; max];
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
std::io::Cursor::new(&data)
|
||||||
|
.take((max as u64) + 1)
|
||||||
|
.read_to_end(&mut buf)
|
||||||
|
.unwrap();
|
||||||
|
// Exactly at cap: buf.len() == max, guard does NOT fire.
|
||||||
|
assert_eq!(buf.len(), max);
|
||||||
|
assert!(buf.len() <= max);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue