1//! Links to "Let Me MeatProxy That For You" (<https://lmmptfy.com/>), made by 2//! CT from the Jev Discord: a site like the old LMGTFY that hands a question 3//! to a chat LLM. Offered where Jev cannot take the question and an LLM would 4//! serve better. 5//! 6//! The link format lives here and nowhere else. It was verified on 7//! 2026-10-05 by reading the site's own JavaScript (fetched raw): one deep 8//! link per provider, in the URL fragment (so it is never sent to a server), 9//! `https://lmmptfy.com/#v3/<provider>?prompt=<encoded>`. The site's decoder 10//! re-encodes what it reads and refuses a mismatch, so `encode` must be 11//! exact: `encodeURIComponent`, then each of `! ' ( ) *` as `%XX` (upper 12//! case), and a final `. ~ _ -` as `%XX` too. The whole fragment may be at 13//! most 7959 characters. A change to the site's format shows up as a 14//! decoder refusing our links; the vectors in the tests came from running 15//! the site's decoder against a JavaScript twin of `encode`. 16//! 17//! The ending also shows what the site's front page looks like when pasted 18//! into Discord or Slack, as a "Provided by" card. The card is built the way 19//! those do: the Worker fetches `BASE` itself (`head`), reads the `<head>`'s 20//! `og:` tags (`extract`), and shows the title, description and the card 21//! image the site chooses. Nothing else is ever fetched: the one origin is 22//! `BASE`; the image is taken only if it is `https` on that same host, at 23//! most `MAX_IMAGE` bytes of `image/*`; the page is read to `MAX_HTML` bytes 24//! at most; both give up after `TIMEOUT_MS`. The image is served from our 25//! own `/meatproxy/card` (`image`) and never linked from the visitor's 26//! browser, which would hand every visitor's address and browser to the 27//! other site. That route takes nothing from the request: it re-derives the 28//! image address from the kept head, so it is no proxy for anything else. 29//! The head is kept an hour per isolate (a failure is not kept, so an outage 30//! is asked again by the next ending), and a failed or empty read shows the 31//! links without the card, never a broken one. No visitor text is ever in a 32//! request or a log line here. 33//! 34//! The parsing and the markup are pure (the tests run natively); only 35//! `head` and `image` touch `worker`. 36 37use std::cell::RefCell; 38 39use axum::body::Body; 40use axum::http::{Response, header}; 41use futures_util::StreamExt; 42use maud::{Markup, PreEscaped, html}; 43 44 45/// The site. 46pub const BASE: &str = "https://lmmptfy.com/"; 47 48/// The one host anything is fetched from. 49const HOST: &str = "lmmptfy.com"; 50 51/// The providers the site has a deep link for: the fragment's name, the text 52/// of the link, and its logo, shuqikhor's pixel icon as vendored (MIT, in 53/// `third-party/pixel-icons`), embedded unchanged (`card::pixel`, which the card is drawn from too). 54pub const PROVIDERS: [(&str, &str, &str); 3] = [ 55 ("claude", "Claude", card::pixel::CLAUDE), 56 ("chatgpt", "ChatGPT", card::pixel::CHATGPT), 57 ("gemini", "Gemini", card::pixel::GEMINI), 58]; 59 60/// What is offered, ahead of the links. 61pub const LEAD: &str = "Please choose an LLM instead:"; 62 63/// What is said above the card. 64pub const PROVIDED: &str = "Provided by:"; 65 66/// How much of the page is read: the `<head>` is first. 67const MAX_HTML: usize = 64 * 1024; 68/// The largest card image taken. 69const MAX_IMAGE: usize = 1_500_000; 70/// How long a fetch, body included, may take. 71const TIMEOUT_MS: u64 = 2_500; 72/// How long an isolate keeps a good head. 73const KEPT_MS: f64 = 3_600_000.0; 74/// Sent with every fetch. 75const AGENT: &str = "lmjtfy-preview (+https://lmjtfy.fun)"; 76/// Our own address for the card image. 77pub const CARD_PATH: &str = "/meatproxy/card"; 78 79const MAX_TITLE: usize = 120; 80const MAX_DESCRIPTION: usize = 300; 81const MAX_SITE: usize = 60; 82 83/// The site's own limit on the fragment, `#` included. 84const SITE_MAX_FRAGMENT: usize = 7959; 85 86/// Ours, well under the site's, so a provider's longer name or a change in 87/// the format cannot push a link over. 88const MAX_FRAGMENT: usize = 7000; 89 90/// `text` as the site's decoder expects a prompt. 91pub fn encode(text: &str) -> String { 92 let mut out = String::with_capacity(text.len() * 3); 93 for byte in text.bytes() { 94 match byte { 95 // `encodeURIComponent` leaves these, and `! ' ( ) *`, alone; 96 // the site wants the last five written out. 97 b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => out.push(byte as char), 98 _ => out.push_str(&format!("%{byte:02X}")), 99 } 100 } 101 // A trailing `. ~ _ -` is written out too. 102 if let Some(last @ ('.' | '~' | '_' | '-')) = out.chars().last() { 103 out.pop(); 104 out.push_str(&format!("%{:02X}", last as u8)); 105 } 106 out 107} 108 109/// The link to `provider` carrying `prompt`, or nothing when the prompt is 110/// empty or would make a fragment the site refuses. It is never shortened: 111/// a cut question is another question. 112pub fn link(provider: &str, prompt: &str) -> Option<String> { 113 if prompt.trim().is_empty() { 114 return None; 115 } 116 let fragment = format!("#v3/{provider}?prompt={}", encode(prompt)); 117 debug_assert!(MAX_FRAGMENT < SITE_MAX_FRAGMENT); 118 (fragment.len() <= MAX_FRAGMENT).then(|| format!("{BASE}{fragment}")) 119} 120 121/// What a page says about itself in its `<head>`, as untrusted text with 122/// its lengths already capped. 123#[derive(Clone, Debug, PartialEq, Eq, Default)] 124pub struct Head { 125 pub title: String, 126 pub description: Option<String>, 127 pub site_name: Option<String>, 128 /// An `https` address on `HOST`, or nothing. 129 pub image: Option<String>, 130} 131 132/// The suggestion, with a link per provider and, if the site's head could be 133/// read, the card of its front page; empty when the question cannot be 134/// carried in a link. 135pub fn suggestion(question: &str, card: Option<&Head>) -> Markup { 136 let links: Vec<(String, &str, &str)> = 137 PROVIDERS.iter().filter_map(|(provider, name, mark)| link(provider, question).map(|href| (href, *name, *mark))).collect(); 138 html! { 139 @if !links.is_empty() { 140 div .meatproxy { 141 p .why { (LEAD) } 142 p .llms { 143 @for (href, name, mark) in &links { 144 a .llm href=(href) target="_blank" rel="noopener noreferrer" { span .llm-mark aria-hidden="true" { (PreEscaped(*mark)) } span { (name) } } 145 } 146 } 147 @if let Some(head) = card { 148 p .why { (PROVIDED) } 149 a .linkcard href=(BASE) target="_blank" rel="noopener noreferrer" { 150 @if head.image.is_some() { img .linkcard-image src=(CARD_PATH) alt="" loading="lazy"; } 151 span .linkcard-text { 152 @if let Some(site) = &head.site_name { span .linkcard-site { (site) } } 153 span .linkcard-title { (head.title) } 154 @if let Some(description) = &head.description { span .linkcard-description { (description) } } 155 span .linkcard-host { (HOST) } 156 } 157 } 158 } 159 } 160 } 161 } 162} 163 164// ---- reading a head ------------------------------------------------------ 165 166/// The head of `html`, the way a link preview reads it: `og:title`, 167/// `og:description` (else `description`), `og:site_name` and `og:image`, 168/// else `<title>`. `None` if there is no title to show. The image is kept 169/// only if it resolves, against `page`, to an `https` address on `HOST`. 170/// Only the first `MAX_HTML` bytes are read. 171pub fn extract(html: &str, page: &str) -> Option<Head> { 172 let html = cut(html, MAX_HTML); 173 let mut metas: Vec<(String, String)> = Vec::new(); 174 let mut title = None; 175 let lower = html.to_ascii_lowercase(); 176 let mut at = 0; 177 while let Some(found) = lower[at..].find('<') { 178 let open = at + found; 179 let rest = &lower[open..]; 180 if rest.starts_with("<!--") { 181 at = lower[open + 4..].find("-->").map_or(lower.len(), |end| open + 4 + end + 3); 182 } else if rest.starts_with("</head") || starts_tag(rest, "<body") { 183 break; 184 } else if starts_tag(rest, "<script") || starts_tag(rest, "<style") { 185 let close = if starts_tag(rest, "<script") { "</script" } else { "</style" }; 186 at = lower[open + 1..].find(close).map_or(lower.len(), |end| open + 1 + end + 1); 187 } else if starts_tag(rest, "<meta") { 188 let end = tag_end(&html[open..]); 189 let attributes = attributes(&html[open + 5..open + end]); 190 let get = |name: &str| attributes.iter().find(|(key, _)| key == name).map(|(_, value)| value.clone()); 191 if let (Some(key), Some(content)) = (get("property").or_else(|| get("name")), get("content")) { 192 metas.push((key.trim().to_ascii_lowercase(), decode(&content))); 193 } 194 at = open + end; 195 } else if starts_tag(rest, "<title") && title.is_none() { 196 let begin = open + tag_end(&html[open..]); 197 let end = lower[begin..].find("</title").map_or(lower.len(), |end| begin + end); 198 title = Some(decode(&html[begin.min(end)..end])); 199 at = end; 200 } else { 201 at = open + 1; 202 } 203 } 204 let first = |key: &str| metas.iter().find(|(name, _)| name == key).map(|(_, value)| tidy(value)).filter(|value| !value.is_empty()); 205 let title = first("og:title").or_else(|| title.map(|title| tidy(&title)).filter(|title| !title.is_empty()))?; 206 Some(Head { 207 title: capped(&title, MAX_TITLE), 208 description: first("og:description").or_else(|| first("description")).map(|text| capped(&text, MAX_DESCRIPTION)), 209 site_name: first("og:site_name").map(|text| capped(&text, MAX_SITE)), 210 image: first("og:image").and_then(|image| same_site(&image, page)), 211 }) 212} 213 214/// `html` cut to at most `max` bytes, on a character boundary. 215fn cut(html: &str, max: usize) -> &str { 216 let mut end = max.min(html.len()); 217 while !html.is_char_boundary(end) { 218 end -= 1; 219 } 220 &html[..end] 221} 222 223/// Does `rest` open the tag `name` (lower case, with its `<`)? 224fn starts_tag(rest: &str, name: &str) -> bool { 225 rest.strip_prefix(name).is_some_and(|after| after.starts_with(|c: char| c.is_ascii_whitespace() || c == '>' || c == '/')) 226} 227 228/// Where the tag at the start of `from` ends, just past its `>`, with quotes 229/// respected; the end of the text if it never does. 230fn tag_end(from: &str) -> usize { 231 let mut quote = None; 232 for (at, c) in from.char_indices() { 233 match (quote, c) { 234 (None, '"' | '\'') => quote = Some(c), 235 (Some(open), c) if c == open => quote = None, 236 (None, '>') => return at + 1, 237 _ => {} 238 } 239 } 240 from.len() 241} 242 243/// A tag's attributes, names in lower case, values as written (undecoded): 244/// double-quoted, single-quoted, bare, or none. 245fn attributes(tag: &str) -> Vec<(String, String)> { 246 let tag = tag.trim_end_matches('>').trim_end_matches('/'); 247 let bytes = tag.as_bytes(); 248 let (mut found, mut at) = (Vec::new(), 0); 249 while at < bytes.len() { 250 while at < bytes.len() && (bytes[at].is_ascii_whitespace() || bytes[at] == b'/') { 251 at += 1; 252 } 253 let begin = at; 254 while at < bytes.len() && !bytes[at].is_ascii_whitespace() && !matches!(bytes[at], b'=' | b'/' | b'>') { 255 at += 1; 256 } 257 let name = tag[begin..at].to_ascii_lowercase(); 258 while at < bytes.len() && bytes[at].is_ascii_whitespace() { 259 at += 1; 260 } 261 let mut value = String::new(); 262 if bytes.get(at) == Some(&b'=') { 263 at += 1; 264 while at < bytes.len() && bytes[at].is_ascii_whitespace() { 265 at += 1; 266 } 267 match bytes.get(at) { 268 Some("e @ (b'"' | b'\'')) => { 269 let end = tag[at + 1..].find(quote as char).map_or(tag.len(), |end| at + 1 + end); 270 value = tag[at + 1..end].to_owned(); 271 at = (end + 1).min(tag.len()); 272 } 273 _ => { 274 let begin = at; 275 while at < bytes.len() && !bytes[at].is_ascii_whitespace() { 276 at += 1; 277 } 278 value = tag[begin..at].to_owned(); 279 } 280 } 281 } 282 if name.is_empty() { 283 at += 1; 284 } else { 285 found.push((name, value)); 286 } 287 } 288 found 289} 290 291/// `text` with the character references HTML text and attributes use read: 292/// the five named ones and numbers, decimal and hex. Anything else stays. 293fn decode(text: &str) -> String { 294 let mut out = String::with_capacity(text.len()); 295 let mut rest = text; 296 while let Some(amp) = rest.find('&') { 297 out.push_str(&rest[..amp]); 298 rest = &rest[amp..]; 299 let read = rest.find(';').filter(|end| *end <= 10).and_then(|end| { 300 let name = &rest[1..end]; 301 let c = match name { 302 "amp" => Some('&'), 303 "lt" => Some('<'), 304 "gt" => Some('>'), 305 "quot" => Some('"'), 306 "apos" => Some('\''), 307 _ => name 308 .strip_prefix('#') 309 .and_then(|n| match n.strip_prefix(['x', 'X']) { 310 Some(hex) => u32::from_str_radix(hex, 16).ok(), 311 None => n.parse().ok(), 312 }) 313 .and_then(char::from_u32), 314 }; 315 c.map(|c| (c, end + 1)) 316 }); 317 match read { 318 Some((c, used)) => { 319 out.push(c); 320 rest = &rest[used..]; 321 } 322 None => { 323 out.push('&'); 324 rest = &rest[1..]; 325 } 326 } 327 } 328 out.push_str(rest); 329 out 330} 331 332/// Runs of whitespace and control characters as one space, trimmed. 333fn tidy(text: &str) -> String { 334 text.split(|c: char| c.is_whitespace() || c.is_control()).filter(|word| !word.is_empty()).collect::<Vec<_>>().join(" ") 335} 336 337/// At most `max` characters, with an ellipsis if it was cut. 338fn capped(text: &str, max: usize) -> String { 339 if text.chars().count() <= max { 340 return text.to_owned(); 341 } 342 let mut cut: String = text.chars().take(max.saturating_sub(1)).collect(); 343 cut.truncate(cut.trim_end().len()); 344 cut.push('…'); 345 cut 346} 347 348/// `image` as an absolute `https` address, resolved against `page`, if and 349/// only if its host is `HOST` itself: no other host, no userinfo, no port. 350fn same_site(image: &str, page: &str) -> Option<String> { 351 let image = image.trim(); 352 if image.is_empty() || image.chars().any(|c| c.is_control() || c.is_whitespace() || c == '\\') { 353 return None; 354 } 355 let absolute = if let Some(rest) = image.strip_prefix("//") { 356 format!("https://{rest}") 357 } else if image.starts_with("https://") { 358 image.to_owned() 359 } else if image.split('/').next().is_some_and(|first| first.contains(':')) { 360 // Another scheme, or `http`. 361 return None; 362 } else if image.starts_with('/') { 363 format!("{}{image}", origin_of(page)?) 364 } else { 365 let base = page.split(['?', '#']).next()?; 366 let directory = &base[..=base.rfind('/')?]; 367 format!("{directory}{}", image.strip_prefix("./").unwrap_or(image)) 368 }; 369 let rest = absolute.strip_prefix("https://")?; 370 let authority = rest.split(['/', '?', '#']).next()?; 371 (authority.eq_ignore_ascii_case(HOST) && !rest[authority.len()..].split(['?', '#']).next()?.split('/').any(|part| part == ".." || part == ".")).then_some(absolute) 372} 373 374/// `https://host` of `page`. 375fn origin_of(page: &str) -> Option<String> { 376 let rest = page.strip_prefix("https://")?; 377 Some(format!("https://{}", rest.split(['/', '?', '#']).next()?)) 378} 379 380// ---- fetching ------------------------------------------------------------ 381 382thread_local! { 383 /// The head, and when it was read: only a good one. 384 static HEAD: RefCell<Option<(f64, Head)>> = const { RefCell::new(None) }; 385 /// The image of that head, by its address, and when it was read. 386 static IMAGE: RefCell<Option<(f64, String, Vec<u8>, String)>> = const { RefCell::new(None) }; 387} 388 389/// The head of `BASE`, kept an hour per isolate. Only a success is kept, so 390/// an outage is asked again by the next ending, and `None` is shown as no 391/// card. Never fails the answer. 392pub async fn head() -> Option<Head> { 393 let now = js_sys::Date::now(); 394 if let Some(kept) = HEAD.with(|kept| kept.borrow().as_ref().filter(|(at, _)| now - at < KEPT_MS).map(|(_, head)| head.clone())) { 395 return Some(kept); 396 } 397 let (_, body) = get(BASE, MAX_HTML, "text/html").await?; 398 let head = extract(&String::from_utf8_lossy(&body), BASE)?; 399 HEAD.with(|kept| *kept.borrow_mut() = Some((now, head.clone()))); 400 Some(head) 401} 402 403/// Where the card image is, if the head names one: the only thing the route 404/// fetches, so with no head there is nothing to serve. 405fn card_address(head: Option<&Head>) -> Option<&str> { 406 head?.image.as_deref() 407} 408 409/// `GET /meatproxy/card`: the card image the site's head names, from our own 410/// address. Nothing in the request is read: the address comes from the head, 411/// which is `BASE`'s and only ever an `https` address on `HOST`. 412pub async fn image() -> Response<Body> { 413 let gone = |status| Response::builder().status(status).header(header::CACHE_CONTROL, "no-store").body(Body::empty()).expect("static headers are valid"); 414 let head = head().await; 415 let Some(url) = card_address(head.as_ref()) else { return gone(404) }; 416 let url = url.to_owned(); 417 let now = js_sys::Date::now(); 418 let kept = IMAGE.with(|kept| kept.borrow().as_ref().filter(|(at, seen, _, _)| now - at < KEPT_MS && *seen == url).map(|(_, _, bytes, kind)| (bytes.clone(), kind.clone()))); 419 let (bytes, kind) = match kept { 420 Some(kept) => kept, 421 None => { 422 let Some((kind, bytes)) = get(&url, MAX_IMAGE, "image/*").await else { return gone(502) }; 423 IMAGE.with(|kept| *kept.borrow_mut() = Some((now, url, bytes.clone(), kind.clone()))); 424 (bytes, kind) 425 } 426 }; 427 Response::builder() 428 .header(header::CONTENT_TYPE, kind) 429 .header("x-content-type-options", "nosniff") 430 .header(header::CACHE_CONTROL, "public, max-age=3600") 431 .body(Body::from(bytes)) 432 .expect("static headers are valid") 433} 434 435/// A GET of `url` (which is `BASE` or an address `same_site` passed), its 436/// content type and at most `max` bytes; `None` for a redirect, a status 437/// other than 200, a type that is not `accept`, a body over `max` (for an 438/// image: an image cut short is no image), or taking over `TIMEOUT_MS`. The 439/// page is cut at `max`, which is what is wanted of a head. 440async fn get(url: &str, max: usize, accept: &str) -> Option<(String, Vec<u8>)> { 441 let work = async { 442 let headers = worker::Headers::new(); 443 headers.set("user-agent", AGENT).ok()?; 444 headers.set("accept", accept).ok()?; 445 let mut init = worker::RequestInit::new(); 446 init.with_headers(headers).with_redirect(worker::RequestRedirect::Manual); 447 let request = worker::Request::new_with_init(url, &init).ok()?; 448 let mut response = worker::Fetch::Request(request).send().await.ok()?; 449 if response.status_code() != 200 { 450 return None; 451 } 452 let kind = response.headers().get("content-type").ok().flatten()?.split(';').next()?.trim().to_ascii_lowercase(); 453 let wanted = match accept.strip_suffix("/*") { 454 Some(group) => kind.starts_with(&format!("{group}/")) && kind.len() > group.len() + 1, 455 None => kind == accept, 456 }; 457 if !wanted { 458 return None; 459 } 460 let truncate = accept != "image/*"; 461 let mut body = Vec::new(); 462 let mut stream = response.stream().ok()?; 463 while let Some(chunk) = stream.next().await { 464 body.extend_from_slice(&chunk.ok()?); 465 if body.len() > max { 466 if truncate { 467 body.truncate(max); 468 break; 469 } 470 return None; 471 } 472 } 473 Some((kind, body)) 474 }; 475 futures_util::pin_mut!(work); 476 let timeout = worker::Delay::from(std::time::Duration::from_millis(TIMEOUT_MS)); 477 match futures_util::future::select(work, timeout).await { 478 futures_util::future::Either::Left((done, _)) => done, 479 futures_util::future::Either::Right(_) => None, 480 } 481} 482 483#[cfg(test)] 484mod tests { 485 use super::*; 486 487 /// Each from the site's own decoder: the JavaScript twin of `encode` 488 /// round-tripped every one (2026-10-05). 489 const VECTORS: [(&str, &str); 6] = [ 490 ("how many sheep do androids dream of?", "how%20many%20sheep%20do%20androids%20dream%20of%3F"), 491 ("what's 2+2 (really)!", "what%27s%202%2B2%20%28really%29%21"), 492 ("ends with dot.", "ends%20with%20dot%2E"), 493 ("café ünïcode 🐑 ~", "caf%C3%A9%20%C3%BCn%C3%AFcode%20%F0%9F%90%91%20%7E"), 494 ("a_b-", "a_b%2D"), 495 ("100% sure & yes #1 ?x=1", "100%25%20sure%20%26%20yes%20%231%20%3Fx%3D1"), 496 ]; 497 498 #[test] 499 fn the_encoder_matches_the_sites_decoder_vectors() { 500 for (text, encoded) in VECTORS { 501 assert_eq!(encode(text), encoded, "{text}"); 502 } 503 } 504 505 #[test] 506 fn an_encoded_prompt_is_only_what_the_decoder_accepts() { 507 for (text, _) in VECTORS { 508 assert!(encode(text).bytes().all(|b| b.is_ascii_alphanumeric() || b"-_.~%".contains(&b)), "{text}"); 509 assert!(!encode(text).ends_with(['.', '~', '_', '-']), "{text}"); 510 } 511 } 512 513 const PAGE: &str = "https://lmmptfy.com/"; 514 515 fn head_of(html: &str) -> Option<Head> { 516 extract(html, PAGE) 517 } 518 519 fn card() -> Head { 520 Head { 521 title: "Let Me MeatProxy that for <You>".into(), 522 description: Some("A \"description\" & more".into()), 523 site_name: Some("LMMPTFY".into()), 524 image: Some("https://lmmptfy.com/assets/og-card-1200x630-XXvjlRvI.png".into()), 525 } 526 } 527 528 #[test] 529 fn three_links_carry_the_question_in_the_fragment() { 530 let html = suggestion("what's 2+2 (really)!", None).into_string(); 531 for (provider, name, _) in PROVIDERS { 532 let href = format!("https://lmmptfy.com/#v3/{provider}?prompt=what%27s%202%2B2%20%28really%29%21"); 533 assert!(html.contains(&format!(r#"<a class="llm" href="{href}" target="_blank" rel="noopener noreferrer">"#)), "{html}"); 534 assert!(html.contains(&format!("<span>{name}</span>")), "{html}"); 535 } 536 assert!(html.contains("Please choose an LLM instead:"), "{html}"); 537 assert_eq!(html.matches(r#"rel="noopener noreferrer""#).count(), 3); 538 assert_eq!(html.matches("<svg").count(), 3); 539 // No card without a head. 540 assert!(!html.contains("Provided by") && !html.contains("linkcard"), "{html}"); 541 } 542 543 #[test] 544 fn a_card_is_built_from_the_head_and_escaped() { 545 let html = suggestion("why?", Some(&card())).into_string(); 546 assert!(html.contains("Provided by:"), "{html}"); 547 assert!(html.contains(r#"<a class="linkcard" href="https://lmmptfy.com/" target="_blank" rel="noopener noreferrer">"#), "{html}"); 548 // Our own address, never the other site's. 549 assert!(html.contains(r#"<img class="linkcard-image" src="/meatproxy/card" alt="" loading="lazy">"#), "{html}"); 550 assert!(!html.contains("og-card"), "{html}"); 551 assert!(html.contains("Let Me MeatProxy that for <You>"), "{html}"); 552 assert!(html.contains("A "description" & more"), "{html}"); 553 assert!(html.contains(r#"<span class="linkcard-host">lmmptfy.com</span>"#), "{html}"); 554 assert_eq!(html.matches(r#"rel="noopener noreferrer""#).count(), 4); 555 } 556 557 #[test] 558 fn a_card_without_an_image_has_no_image_element() { 559 let head = Head { image: None, description: None, site_name: None, ..card() }; 560 let html = suggestion("why?", Some(&head)).into_string(); 561 assert!(html.contains("linkcard-title") && !html.contains("<img") && !html.contains("linkcard-description"), "{html}"); 562 } 563 564 #[test] 565 fn an_overlong_or_blank_question_gets_no_links() { 566 assert!(suggestion(&"a".repeat(6_900), None).into_string().contains("lmmptfy.com/#v3/claude")); 567 assert_eq!(suggestion(&"a".repeat(7_000), Some(&card())).into_string(), ""); 568 // Two bytes each, six encoded characters each. 569 assert_eq!(suggestion(&"é".repeat(1_200), None).into_string(), ""); 570 assert_eq!(suggestion(" ", Some(&card())).into_string(), ""); 571 } 572 573 #[test] 574 fn the_logos_are_all_there_distinct_and_in_their_upstream_colours() { 575 let marks: Vec<_> = PROVIDERS.iter().map(|(_, _, svg)| *svg).collect(); 576 for (mark, class) in marks.iter().zip(["pixelicon-claude", "pixelicon-chatgpt", "pixelicon-gemini"]) { 577 assert!(mark.starts_with("<svg") && mark.contains("viewBox=\"0 0 9 9\"") && mark.contains(class), "{mark}"); 578 } 579 for (a, b) in [(0, 1), (0, 2), (1, 2)] { 580 assert_ne!(marks[a], marks[b]); 581 } 582 let html = suggestion("why?", None).into_string(); 583 for colour in ["#D97757", "#343341", "#7C72C6", "#637FCE", "#9168C0", "#438DD7", "#1BA1E3"] { 584 assert!(html.contains(&format!("fill=\"{colour}\"")), "{colour}"); 585 } 586 // Not recoloured, and the caption names each. 587 assert!(!html.contains("currentColor")); 588 } 589 590 #[test] 591 fn og_tags_in_any_order_and_quote_style() { 592 let head = head_of( 593 r#"<!doctype html><html><head><META CONTENT='Let Me & MeatProxy' Property="og:title"> 594 <meta name=description content="plain "desc""> 595 <meta content="https://lmmptfy.com/a.png?x=1&y=2" property="og:image" /> 596 <meta property='og:site_name' content='LMMPTFY'></head><body><meta property="og:title" content="late"></body>"#, 597 ) 598 .unwrap(); 599 assert_eq!(head.title, "Let Me & MeatProxy"); 600 assert_eq!(head.description.as_deref(), Some(r#"plain "desc""#)); 601 assert_eq!(head.site_name.as_deref(), Some("LMMPTFY")); 602 assert_eq!(head.image.as_deref(), Some("https://lmmptfy.com/a.png?x=1&y=2")); 603 } 604 605 #[test] 606 fn og_description_beats_the_name_one_and_numeric_entities_decode() { 607 let head = head_of(r#"<head><meta name="description" content="b"><meta property="og:description" content="a 'q' 🐑 &bogus; &"><meta property="og:title" content="t"></head>"#).unwrap(); 608 assert_eq!(head.description.as_deref(), Some("a 'q' 🐑 &bogus; &")); 609 } 610 611 #[test] 612 fn a_missing_og_title_falls_back_to_the_title_element() { 613 let head = head_of("<head><title>\n Hello\n there & you </title><meta property=\"og:image\" content=\"/assets/c.png\"></head>").unwrap(); 614 assert_eq!(head.title, "Hello there & you"); 615 assert_eq!(head.description, None); 616 assert_eq!(head.image.as_deref(), Some("https://lmmptfy.com/assets/c.png")); 617 } 618 619 #[test] 620 fn a_relative_image_is_resolved_against_the_page() { 621 let at = |image: &str| same_site(image, "https://lmmptfy.com/deep/page.html?x#y"); 622 assert_eq!(at("/a.png").as_deref(), Some("https://lmmptfy.com/a.png")); 623 assert_eq!(at("a.png").as_deref(), Some("https://lmmptfy.com/deep/a.png")); 624 assert_eq!(at("./a.png").as_deref(), Some("https://lmmptfy.com/deep/a.png")); 625 assert_eq!(at("//lmmptfy.com/a.png").as_deref(), Some("https://lmmptfy.com/a.png")); 626 assert_eq!(at("https://LMMPTFY.com/a.png").as_deref(), Some("https://LMMPTFY.com/a.png")); 627 } 628 629 #[test] 630 fn an_image_off_the_site_is_dropped_and_the_card_kept() { 631 for image in [ 632 "https://evil.example/a.png", 633 "http://lmmptfy.com/a.png", 634 "//evil.example/a.png", 635 "https://lmmptfy.com.evil.example/a.png", 636 "https://evil.example/https://lmmptfy.com/a.png", 637 "https://lmmptfy.com@evil.example/a.png", 638 "https://evil.example@lmmptfy.com/a.png", 639 "https://lmmptfy.com:8443/a.png", 640 "https://lmmptfy.com/../a.png", 641 "https://lmmptfy.com\\@evil.example/a.png", 642 "data:image/png;base64,AAAA", 643 "javascript:alert(1)", 644 "ftp://lmmptfy.com/a.png", 645 "", 646 ] { 647 let html = format!(r#"<head><meta property="og:title" content="T"><meta property="og:image" content="{}"></head>"#, image.replace('"', """)); 648 let head = head_of(&html).unwrap(); 649 assert_eq!(head.image, None, "{image}"); 650 assert_eq!(head.title, "T"); 651 } 652 } 653 654 #[test] 655 fn a_huge_page_is_read_only_to_the_cap_and_text_is_capped() { 656 let wanted = format!(r#"<head><meta property="og:title" content="{}"><meta property="og:description" content="{}">"#, "t".repeat(500), "d".repeat(5_000)); 657 let late = format!("{wanted}{}<meta property=\"og:site_name\" content=\"too late\"></head>", " ".repeat(MAX_HTML)); 658 let head = head_of(&late).unwrap(); 659 assert_eq!(head.site_name, None); 660 assert_eq!(head.title.chars().count(), MAX_TITLE); 661 assert!(head.title.ends_with('…')); 662 assert_eq!(head.description.unwrap().chars().count(), MAX_DESCRIPTION); 663 // A cut inside a multibyte character does not panic. 664 let wide = format!("<head><title>x</title>{}</head>", "é".repeat(MAX_HTML)); 665 assert_eq!(head_of(&wide).unwrap().title, "x"); 666 } 667 668 #[test] 669 fn comments_scripts_and_the_body_are_not_the_head() { 670 let head = head_of( 671 r#"<head><!-- <meta property="og:title" content="comment"> --><script>var s = '<meta property="og:title" content="script">';</script> 672 <style>/* <title>style</title> */</style><meta property="og:title" content="real"></head>"#, 673 ) 674 .unwrap(); 675 assert_eq!(head.title, "real"); 676 } 677 678 #[test] 679 fn the_route_has_nothing_to_serve_without_a_head_or_an_image() { 680 assert_eq!(card_address(None), None); 681 assert_eq!(card_address(Some(&Head { image: None, ..card() })), None); 682 assert_eq!(card_address(Some(&card())), Some("https://lmmptfy.com/assets/og-card-1200x630-XXvjlRvI.png")); 683 } 684 685 #[test] 686 fn nothing_to_show_is_none() { 687 assert_eq!(head_of(""), None); 688 assert_eq!(head_of("<head><meta charset=utf-8><meta name=viewport content='width=1'></head>"), None); 689 assert_eq!(head_of(r#"<head><meta property="og:image" content="/a.png"><title> </title></head>"#), None); 690 assert_eq!(head_of("<head><meta property=og:title content="), None); 691 assert_eq!(head_of("<meta <<<>>> =\"= <title"), None); 692 } 693 694 /// Needs the network: the site's real front page, read as the Worker 695 /// would. Run with `cargo test -p lmjtfy --lib -- --ignored real_site --nocapture`. 696 #[test] 697 #[ignore] 698 fn real_site_head() { 699 let out = std::process::Command::new("curl").args(["-sSL", "-A", AGENT, "--max-time", "10", BASE]).output().expect("curl"); 700 let html = String::from_utf8_lossy(&out.stdout); 701 let head = extract(&html, BASE).expect("a head"); 702 println!("{head:#?}"); 703 assert!(!head.title.is_empty()); 704 assert!(head.image.as_deref().is_some_and(|image| image.starts_with("https://lmmptfy.com/"))); 705 } 706}