lmjtfy.git / apps / lmjtfy / src / meatproxy.rs
1//! Links to "Let Me MeatProxy That For You" (<https://lmmptfy.com/>), made by
2//! CT from the Jev Discord: a site like the old LMGTFY that hands a question
3//! to a chat LLM. Offered where Jev cannot take the question and an LLM would
4//! serve better.
5//!
6//! The link format lives here and nowhere else. It was verified on
7//! 2026-10-05 by reading the site's own JavaScript (fetched raw): one deep
8//! link per provider, in the URL fragment (so it is never sent to a server),
9//! `https://lmmptfy.com/#v3/<provider>?prompt=<encoded>`. The site's decoder
10//! re-encodes what it reads and refuses a mismatch, so `encode` must be
11//! exact: `encodeURIComponent`, then each of `! ' ( ) *` as `%XX` (upper
12//! case), and a final `. ~ _ -` as `%XX` too. The whole fragment may be at
13//! most 7959 characters. A change to the site's format shows up as a
14//! decoder refusing our links; the vectors in the tests came from running
15//! the site's decoder against a JavaScript twin of `encode`.
16//!
17//! The ending also shows what the site's front page looks like when pasted
18//! into Discord or Slack, as a "Provided by" card. The card is built the way
19//! those do: the Worker fetches `BASE` itself (`head`), reads the `<head>`'s
20//! `og:` tags (`extract`), and shows the title, description and the card
21//! image the site chooses. Nothing else is ever fetched: the one origin is
22//! `BASE`; the image is taken only if it is `https` on that same host, at
23//! most `MAX_IMAGE` bytes of `image/*`; the page is read to `MAX_HTML` bytes
24//! at most; both give up after `TIMEOUT_MS`. The image is served from our
25//! own `/meatproxy/card` (`image`) and never linked from the visitor's
26//! browser, which would hand every visitor's address and browser to the
27//! other site. That route takes nothing from the request: it re-derives the
28//! image address from the kept head, so it is no proxy for anything else.
29//! The head is kept an hour per isolate (a failure is not kept, so an outage
30//! is asked again by the next ending), and a failed or empty read shows the
31//! links without the card, never a broken one. No visitor text is ever in a
32//! request or a log line here.
33//!
34//! The parsing and the markup are pure (the tests run natively); only
35//! `head` and `image` touch `worker`.
36
37use std::cell::RefCell;
38
39use axum::body::Body;
40use axum::http::{Response, header};
41use futures_util::StreamExt;
42use maud::{Markup, PreEscaped, html};
43
44
45/// The site.
46pub const BASE: &str = "https://lmmptfy.com/";
47
48/// The one host anything is fetched from.
49const HOST: &str = "lmmptfy.com";
50
51/// The providers the site has a deep link for: the fragment's name, the text
52/// of the link, and its logo, shuqikhor's pixel icon as vendored (MIT, in
53/// `third-party/pixel-icons`), embedded unchanged (`card::pixel`, which the card is drawn from too).
54pub const PROVIDERS: [(&str, &str, &str); 3] = [
55    ("claude", "Claude", card::pixel::CLAUDE),
56    ("chatgpt", "ChatGPT", card::pixel::CHATGPT),
57    ("gemini", "Gemini", card::pixel::GEMINI),
58];
59
60/// What is offered, ahead of the links.
61pub const LEAD: &str = "Please choose an LLM instead:";
62
63/// What is said above the card.
64pub const PROVIDED: &str = "Provided by:";
65
66/// How much of the page is read: the `<head>` is first.
67const MAX_HTML: usize = 64 * 1024;
68/// The largest card image taken.
69const MAX_IMAGE: usize = 1_500_000;
70/// How long a fetch, body included, may take.
71const TIMEOUT_MS: u64 = 2_500;
72/// How long an isolate keeps a good head.
73const KEPT_MS: f64 = 3_600_000.0;
74/// Sent with every fetch.
75const AGENT: &str = "lmjtfy-preview (+https://lmjtfy.fun)";
76/// Our own address for the card image.
77pub const CARD_PATH: &str = "/meatproxy/card";
78
79const MAX_TITLE: usize = 120;
80const MAX_DESCRIPTION: usize = 300;
81const MAX_SITE: usize = 60;
82
83/// The site's own limit on the fragment, `#` included.
84const SITE_MAX_FRAGMENT: usize = 7959;
85
86/// Ours, well under the site's, so a provider's longer name or a change in
87/// the format cannot push a link over.
88const MAX_FRAGMENT: usize = 7000;
89
90/// `text` as the site's decoder expects a prompt.
91pub fn encode(text: &str) -> String {
92    let mut out = String::with_capacity(text.len() * 3);
93    for byte in text.bytes() {
94        match byte {
95            // `encodeURIComponent` leaves these, and `! ' ( ) *`, alone;
96            // the site wants the last five written out.
97            b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => out.push(byte as char),
98            _ => out.push_str(&format!("%{byte:02X}")),
99        }
100    }
101    // A trailing `. ~ _ -` is written out too.
102    if let Some(last @ ('.' | '~' | '_' | '-')) = out.chars().last() {
103        out.pop();
104        out.push_str(&format!("%{:02X}", last as u8));
105    }
106    out
107}
108
109/// The link to `provider` carrying `prompt`, or nothing when the prompt is
110/// empty or would make a fragment the site refuses. It is never shortened:
111/// a cut question is another question.
112pub fn link(provider: &str, prompt: &str) -> Option<String> {
113    if prompt.trim().is_empty() {
114        return None;
115    }
116    let fragment = format!("#v3/{provider}?prompt={}", encode(prompt));
117    debug_assert!(MAX_FRAGMENT < SITE_MAX_FRAGMENT);
118    (fragment.len() <= MAX_FRAGMENT).then(|| format!("{BASE}{fragment}"))
119}
120
121/// What a page says about itself in its `<head>`, as untrusted text with
122/// its lengths already capped.
123#[derive(Clone, Debug, PartialEq, Eq, Default)]
124pub struct Head {
125    pub title: String,
126    pub description: Option<String>,
127    pub site_name: Option<String>,
128    /// An `https` address on `HOST`, or nothing.
129    pub image: Option<String>,
130}
131
132/// The suggestion, with a link per provider and, if the site's head could be
133/// read, the card of its front page; empty when the question cannot be
134/// carried in a link.
135pub fn suggestion(question: &str, card: Option<&Head>) -> Markup {
136    let links: Vec<(String, &str, &str)> =
137        PROVIDERS.iter().filter_map(|(provider, name, mark)| link(provider, question).map(|href| (href, *name, *mark))).collect();
138    html! {
139        @if !links.is_empty() {
140            div .meatproxy {
141                p .why { (LEAD) }
142                p .llms {
143                    @for (href, name, mark) in &links {
144                        a .llm href=(href) target="_blank" rel="noopener noreferrer" { span .llm-mark aria-hidden="true" { (PreEscaped(*mark)) } span { (name) } }
145                    }
146                }
147                @if let Some(head) = card {
148                    p .why { (PROVIDED) }
149                    a .linkcard href=(BASE) target="_blank" rel="noopener noreferrer" {
150                        @if head.image.is_some() { img .linkcard-image src=(CARD_PATH) alt="" loading="lazy"; }
151                        span .linkcard-text {
152                            @if let Some(site) = &head.site_name { span .linkcard-site { (site) } }
153                            span .linkcard-title { (head.title) }
154                            @if let Some(description) = &head.description { span .linkcard-description { (description) } }
155                            span .linkcard-host { (HOST) }
156                        }
157                    }
158                }
159            }
160        }
161    }
162}
163
164// ---- reading a head ------------------------------------------------------
165
166/// The head of `html`, the way a link preview reads it: `og:title`,
167/// `og:description` (else `description`), `og:site_name` and `og:image`,
168/// else `<title>`. `None` if there is no title to show. The image is kept
169/// only if it resolves, against `page`, to an `https` address on `HOST`.
170/// Only the first `MAX_HTML` bytes are read.
171pub fn extract(html: &str, page: &str) -> Option<Head> {
172    let html = cut(html, MAX_HTML);
173    let mut metas: Vec<(String, String)> = Vec::new();
174    let mut title = None;
175    let lower = html.to_ascii_lowercase();
176    let mut at = 0;
177    while let Some(found) = lower[at..].find('<') {
178        let open = at + found;
179        let rest = &lower[open..];
180        if rest.starts_with("<!--") {
181            at = lower[open + 4..].find("-->").map_or(lower.len(), |end| open + 4 + end + 3);
182        } else if rest.starts_with("</head") || starts_tag(rest, "<body") {
183            break;
184        } else if starts_tag(rest, "<script") || starts_tag(rest, "<style") {
185            let close = if starts_tag(rest, "<script") { "</script" } else { "</style" };
186            at = lower[open + 1..].find(close).map_or(lower.len(), |end| open + 1 + end + 1);
187        } else if starts_tag(rest, "<meta") {
188            let end = tag_end(&html[open..]);
189            let attributes = attributes(&html[open + 5..open + end]);
190            let get = |name: &str| attributes.iter().find(|(key, _)| key == name).map(|(_, value)| value.clone());
191            if let (Some(key), Some(content)) = (get("property").or_else(|| get("name")), get("content")) {
192                metas.push((key.trim().to_ascii_lowercase(), decode(&content)));
193            }
194            at = open + end;
195        } else if starts_tag(rest, "<title") && title.is_none() {
196            let begin = open + tag_end(&html[open..]);
197            let end = lower[begin..].find("</title").map_or(lower.len(), |end| begin + end);
198            title = Some(decode(&html[begin.min(end)..end]));
199            at = end;
200        } else {
201            at = open + 1;
202        }
203    }
204    let first = |key: &str| metas.iter().find(|(name, _)| name == key).map(|(_, value)| tidy(value)).filter(|value| !value.is_empty());
205    let title = first("og:title").or_else(|| title.map(|title| tidy(&title)).filter(|title| !title.is_empty()))?;
206    Some(Head {
207        title: capped(&title, MAX_TITLE),
208        description: first("og:description").or_else(|| first("description")).map(|text| capped(&text, MAX_DESCRIPTION)),
209        site_name: first("og:site_name").map(|text| capped(&text, MAX_SITE)),
210        image: first("og:image").and_then(|image| same_site(&image, page)),
211    })
212}
213
214/// `html` cut to at most `max` bytes, on a character boundary.
215fn cut(html: &str, max: usize) -> &str {
216    let mut end = max.min(html.len());
217    while !html.is_char_boundary(end) {
218        end -= 1;
219    }
220    &html[..end]
221}
222
223/// Does `rest` open the tag `name` (lower case, with its `<`)?
224fn starts_tag(rest: &str, name: &str) -> bool {
225    rest.strip_prefix(name).is_some_and(|after| after.starts_with(|c: char| c.is_ascii_whitespace() || c == '>' || c == '/'))
226}
227
228/// Where the tag at the start of `from` ends, just past its `>`, with quotes
229/// respected; the end of the text if it never does.
230fn tag_end(from: &str) -> usize {
231    let mut quote = None;
232    for (at, c) in from.char_indices() {
233        match (quote, c) {
234            (None, '"' | '\'') => quote = Some(c),
235            (Some(open), c) if c == open => quote = None,
236            (None, '>') => return at + 1,
237            _ => {}
238        }
239    }
240    from.len()
241}
242
243/// A tag's attributes, names in lower case, values as written (undecoded):
244/// double-quoted, single-quoted, bare, or none.
245fn attributes(tag: &str) -> Vec<(String, String)> {
246    let tag = tag.trim_end_matches('>').trim_end_matches('/');
247    let bytes = tag.as_bytes();
248    let (mut found, mut at) = (Vec::new(), 0);
249    while at < bytes.len() {
250        while at < bytes.len() && (bytes[at].is_ascii_whitespace() || bytes[at] == b'/') {
251            at += 1;
252        }
253        let begin = at;
254        while at < bytes.len() && !bytes[at].is_ascii_whitespace() && !matches!(bytes[at], b'=' | b'/' | b'>') {
255            at += 1;
256        }
257        let name = tag[begin..at].to_ascii_lowercase();
258        while at < bytes.len() && bytes[at].is_ascii_whitespace() {
259            at += 1;
260        }
261        let mut value = String::new();
262        if bytes.get(at) == Some(&b'=') {
263            at += 1;
264            while at < bytes.len() && bytes[at].is_ascii_whitespace() {
265                at += 1;
266            }
267            match bytes.get(at) {
268                Some(&quote @ (b'"' | b'\'')) => {
269                    let end = tag[at + 1..].find(quote as char).map_or(tag.len(), |end| at + 1 + end);
270                    value = tag[at + 1..end].to_owned();
271                    at = (end + 1).min(tag.len());
272                }
273                _ => {
274                    let begin = at;
275                    while at < bytes.len() && !bytes[at].is_ascii_whitespace() {
276                        at += 1;
277                    }
278                    value = tag[begin..at].to_owned();
279                }
280            }
281        }
282        if name.is_empty() {
283            at += 1;
284        } else {
285            found.push((name, value));
286        }
287    }
288    found
289}
290
291/// `text` with the character references HTML text and attributes use read:
292/// the five named ones and numbers, decimal and hex. Anything else stays.
293fn decode(text: &str) -> String {
294    let mut out = String::with_capacity(text.len());
295    let mut rest = text;
296    while let Some(amp) = rest.find('&') {
297        out.push_str(&rest[..amp]);
298        rest = &rest[amp..];
299        let read = rest.find(';').filter(|end| *end <= 10).and_then(|end| {
300            let name = &rest[1..end];
301            let c = match name {
302                "amp" => Some('&'),
303                "lt" => Some('<'),
304                "gt" => Some('>'),
305                "quot" => Some('"'),
306                "apos" => Some('\''),
307                _ => name
308                    .strip_prefix('#')
309                    .and_then(|n| match n.strip_prefix(['x', 'X']) {
310                        Some(hex) => u32::from_str_radix(hex, 16).ok(),
311                        None => n.parse().ok(),
312                    })
313                    .and_then(char::from_u32),
314            };
315            c.map(|c| (c, end + 1))
316        });
317        match read {
318            Some((c, used)) => {
319                out.push(c);
320                rest = &rest[used..];
321            }
322            None => {
323                out.push('&');
324                rest = &rest[1..];
325            }
326        }
327    }
328    out.push_str(rest);
329    out
330}
331
332/// Runs of whitespace and control characters as one space, trimmed.
333fn tidy(text: &str) -> String {
334    text.split(|c: char| c.is_whitespace() || c.is_control()).filter(|word| !word.is_empty()).collect::<Vec<_>>().join(" ")
335}
336
337/// At most `max` characters, with an ellipsis if it was cut.
338fn capped(text: &str, max: usize) -> String {
339    if text.chars().count() <= max {
340        return text.to_owned();
341    }
342    let mut cut: String = text.chars().take(max.saturating_sub(1)).collect();
343    cut.truncate(cut.trim_end().len());
344    cut.push('…');
345    cut
346}
347
348/// `image` as an absolute `https` address, resolved against `page`, if and
349/// only if its host is `HOST` itself: no other host, no userinfo, no port.
350fn same_site(image: &str, page: &str) -> Option<String> {
351    let image = image.trim();
352    if image.is_empty() || image.chars().any(|c| c.is_control() || c.is_whitespace() || c == '\\') {
353        return None;
354    }
355    let absolute = if let Some(rest) = image.strip_prefix("//") {
356        format!("https://{rest}")
357    } else if image.starts_with("https://") {
358        image.to_owned()
359    } else if image.split('/').next().is_some_and(|first| first.contains(':')) {
360        // Another scheme, or `http`.
361        return None;
362    } else if image.starts_with('/') {
363        format!("{}{image}", origin_of(page)?)
364    } else {
365        let base = page.split(['?', '#']).next()?;
366        let directory = &base[..=base.rfind('/')?];
367        format!("{directory}{}", image.strip_prefix("./").unwrap_or(image))
368    };
369    let rest = absolute.strip_prefix("https://")?;
370    let authority = rest.split(['/', '?', '#']).next()?;
371    (authority.eq_ignore_ascii_case(HOST) && !rest[authority.len()..].split(['?', '#']).next()?.split('/').any(|part| part == ".." || part == ".")).then_some(absolute)
372}
373
374/// `https://host` of `page`.
375fn origin_of(page: &str) -> Option<String> {
376    let rest = page.strip_prefix("https://")?;
377    Some(format!("https://{}", rest.split(['/', '?', '#']).next()?))
378}
379
380// ---- fetching ------------------------------------------------------------
381
382thread_local! {
383    /// The head, and when it was read: only a good one.
384    static HEAD: RefCell<Option<(f64, Head)>> = const { RefCell::new(None) };
385    /// The image of that head, by its address, and when it was read.
386    static IMAGE: RefCell<Option<(f64, String, Vec<u8>, String)>> = const { RefCell::new(None) };
387}
388
389/// The head of `BASE`, kept an hour per isolate. Only a success is kept, so
390/// an outage is asked again by the next ending, and `None` is shown as no
391/// card. Never fails the answer.
392pub async fn head() -> Option<Head> {
393    let now = js_sys::Date::now();
394    if let Some(kept) = HEAD.with(|kept| kept.borrow().as_ref().filter(|(at, _)| now - at < KEPT_MS).map(|(_, head)| head.clone())) {
395        return Some(kept);
396    }
397    let (_, body) = get(BASE, MAX_HTML, "text/html").await?;
398    let head = extract(&String::from_utf8_lossy(&body), BASE)?;
399    HEAD.with(|kept| *kept.borrow_mut() = Some((now, head.clone())));
400    Some(head)
401}
402
403/// Where the card image is, if the head names one: the only thing the route
404/// fetches, so with no head there is nothing to serve.
405fn card_address(head: Option<&Head>) -> Option<&str> {
406    head?.image.as_deref()
407}
408
409/// `GET /meatproxy/card`: the card image the site's head names, from our own
410/// address. Nothing in the request is read: the address comes from the head,
411/// which is `BASE`'s and only ever an `https` address on `HOST`.
412pub async fn image() -> Response<Body> {
413    let gone = |status| Response::builder().status(status).header(header::CACHE_CONTROL, "no-store").body(Body::empty()).expect("static headers are valid");
414    let head = head().await;
415    let Some(url) = card_address(head.as_ref()) else { return gone(404) };
416    let url = url.to_owned();
417    let now = js_sys::Date::now();
418    let kept = IMAGE.with(|kept| kept.borrow().as_ref().filter(|(at, seen, _, _)| now - at < KEPT_MS && *seen == url).map(|(_, _, bytes, kind)| (bytes.clone(), kind.clone())));
419    let (bytes, kind) = match kept {
420        Some(kept) => kept,
421        None => {
422            let Some((kind, bytes)) = get(&url, MAX_IMAGE, "image/*").await else { return gone(502) };
423            IMAGE.with(|kept| *kept.borrow_mut() = Some((now, url, bytes.clone(), kind.clone())));
424            (bytes, kind)
425        }
426    };
427    Response::builder()
428        .header(header::CONTENT_TYPE, kind)
429        .header("x-content-type-options", "nosniff")
430        .header(header::CACHE_CONTROL, "public, max-age=3600")
431        .body(Body::from(bytes))
432        .expect("static headers are valid")
433}
434
435/// A GET of `url` (which is `BASE` or an address `same_site` passed), its
436/// content type and at most `max` bytes; `None` for a redirect, a status
437/// other than 200, a type that is not `accept`, a body over `max` (for an
438/// image: an image cut short is no image), or taking over `TIMEOUT_MS`. The
439/// page is cut at `max`, which is what is wanted of a head.
440async fn get(url: &str, max: usize, accept: &str) -> Option<(String, Vec<u8>)> {
441    let work = async {
442        let headers = worker::Headers::new();
443        headers.set("user-agent", AGENT).ok()?;
444        headers.set("accept", accept).ok()?;
445        let mut init = worker::RequestInit::new();
446        init.with_headers(headers).with_redirect(worker::RequestRedirect::Manual);
447        let request = worker::Request::new_with_init(url, &init).ok()?;
448        let mut response = worker::Fetch::Request(request).send().await.ok()?;
449        if response.status_code() != 200 {
450            return None;
451        }
452        let kind = response.headers().get("content-type").ok().flatten()?.split(';').next()?.trim().to_ascii_lowercase();
453        let wanted = match accept.strip_suffix("/*") {
454            Some(group) => kind.starts_with(&format!("{group}/")) && kind.len() > group.len() + 1,
455            None => kind == accept,
456        };
457        if !wanted {
458            return None;
459        }
460        let truncate = accept != "image/*";
461        let mut body = Vec::new();
462        let mut stream = response.stream().ok()?;
463        while let Some(chunk) = stream.next().await {
464            body.extend_from_slice(&chunk.ok()?);
465            if body.len() > max {
466                if truncate {
467                    body.truncate(max);
468                    break;
469                }
470                return None;
471            }
472        }
473        Some((kind, body))
474    };
475    futures_util::pin_mut!(work);
476    let timeout = worker::Delay::from(std::time::Duration::from_millis(TIMEOUT_MS));
477    match futures_util::future::select(work, timeout).await {
478        futures_util::future::Either::Left((done, _)) => done,
479        futures_util::future::Either::Right(_) => None,
480    }
481}
482
483#[cfg(test)]
484mod tests {
485    use super::*;
486
487    /// Each from the site's own decoder: the JavaScript twin of `encode`
488    /// round-tripped every one (2026-10-05).
489    const VECTORS: [(&str, &str); 6] = [
490        ("how many sheep do androids dream of?", "how%20many%20sheep%20do%20androids%20dream%20of%3F"),
491        ("what's 2+2 (really)!", "what%27s%202%2B2%20%28really%29%21"),
492        ("ends with dot.", "ends%20with%20dot%2E"),
493        ("café ünïcode 🐑 ~", "caf%C3%A9%20%C3%BCn%C3%AFcode%20%F0%9F%90%91%20%7E"),
494        ("a_b-", "a_b%2D"),
495        ("100% sure & yes #1 ?x=1", "100%25%20sure%20%26%20yes%20%231%20%3Fx%3D1"),
496    ];
497
498    #[test]
499    fn the_encoder_matches_the_sites_decoder_vectors() {
500        for (text, encoded) in VECTORS {
501            assert_eq!(encode(text), encoded, "{text}");
502        }
503    }
504
505    #[test]
506    fn an_encoded_prompt_is_only_what_the_decoder_accepts() {
507        for (text, _) in VECTORS {
508            assert!(encode(text).bytes().all(|b| b.is_ascii_alphanumeric() || b"-_.~%".contains(&b)), "{text}");
509            assert!(!encode(text).ends_with(['.', '~', '_', '-']), "{text}");
510        }
511    }
512
513    const PAGE: &str = "https://lmmptfy.com/";
514
515    fn head_of(html: &str) -> Option<Head> {
516        extract(html, PAGE)
517    }
518
519    fn card() -> Head {
520        Head {
521            title: "Let Me MeatProxy that for <You>".into(),
522            description: Some("A \"description\" & more".into()),
523            site_name: Some("LMMPTFY".into()),
524            image: Some("https://lmmptfy.com/assets/og-card-1200x630-XXvjlRvI.png".into()),
525        }
526    }
527
528    #[test]
529    fn three_links_carry_the_question_in_the_fragment() {
530        let html = suggestion("what's 2+2 (really)!", None).into_string();
531        for (provider, name, _) in PROVIDERS {
532            let href = format!("https://lmmptfy.com/#v3/{provider}?prompt=what%27s%202%2B2%20%28really%29%21");
533            assert!(html.contains(&format!(r#"<a class="llm" href="{href}" target="_blank" rel="noopener noreferrer">"#)), "{html}");
534            assert!(html.contains(&format!("<span>{name}</span>")), "{html}");
535        }
536        assert!(html.contains("Please choose an LLM instead:"), "{html}");
537        assert_eq!(html.matches(r#"rel="noopener noreferrer""#).count(), 3);
538        assert_eq!(html.matches("<svg").count(), 3);
539        // No card without a head.
540        assert!(!html.contains("Provided by") && !html.contains("linkcard"), "{html}");
541    }
542
543    #[test]
544    fn a_card_is_built_from_the_head_and_escaped() {
545        let html = suggestion("why?", Some(&card())).into_string();
546        assert!(html.contains("Provided by:"), "{html}");
547        assert!(html.contains(r#"<a class="linkcard" href="https://lmmptfy.com/" target="_blank" rel="noopener noreferrer">"#), "{html}");
548        // Our own address, never the other site's.
549        assert!(html.contains(r#"<img class="linkcard-image" src="/meatproxy/card" alt="" loading="lazy">"#), "{html}");
550        assert!(!html.contains("og-card"), "{html}");
551        assert!(html.contains("Let Me MeatProxy that for &lt;You&gt;"), "{html}");
552        assert!(html.contains("A &quot;description&quot; &amp; more"), "{html}");
553        assert!(html.contains(r#"<span class="linkcard-host">lmmptfy.com</span>"#), "{html}");
554        assert_eq!(html.matches(r#"rel="noopener noreferrer""#).count(), 4);
555    }
556
557    #[test]
558    fn a_card_without_an_image_has_no_image_element() {
559        let head = Head { image: None, description: None, site_name: None, ..card() };
560        let html = suggestion("why?", Some(&head)).into_string();
561        assert!(html.contains("linkcard-title") && !html.contains("<img") && !html.contains("linkcard-description"), "{html}");
562    }
563
564    #[test]
565    fn an_overlong_or_blank_question_gets_no_links() {
566        assert!(suggestion(&"a".repeat(6_900), None).into_string().contains("lmmptfy.com/#v3/claude"));
567        assert_eq!(suggestion(&"a".repeat(7_000), Some(&card())).into_string(), "");
568        // Two bytes each, six encoded characters each.
569        assert_eq!(suggestion(&"é".repeat(1_200), None).into_string(), "");
570        assert_eq!(suggestion("   ", Some(&card())).into_string(), "");
571    }
572
573    #[test]
574    fn the_logos_are_all_there_distinct_and_in_their_upstream_colours() {
575        let marks: Vec<_> = PROVIDERS.iter().map(|(_, _, svg)| *svg).collect();
576        for (mark, class) in marks.iter().zip(["pixelicon-claude", "pixelicon-chatgpt", "pixelicon-gemini"]) {
577            assert!(mark.starts_with("<svg") && mark.contains("viewBox=\"0 0 9 9\"") && mark.contains(class), "{mark}");
578        }
579        for (a, b) in [(0, 1), (0, 2), (1, 2)] {
580            assert_ne!(marks[a], marks[b]);
581        }
582        let html = suggestion("why?", None).into_string();
583        for colour in ["#D97757", "#343341", "#7C72C6", "#637FCE", "#9168C0", "#438DD7", "#1BA1E3"] {
584            assert!(html.contains(&format!("fill=\"{colour}\"")), "{colour}");
585        }
586        // Not recoloured, and the caption names each.
587        assert!(!html.contains("currentColor"));
588    }
589
590    #[test]
591    fn og_tags_in_any_order_and_quote_style() {
592        let head = head_of(
593            r#"<!doctype html><html><head><META CONTENT='Let Me &amp; MeatProxy' Property="og:title">
594            <meta name=description content="plain &quot;desc&quot;">
595            <meta content="https://lmmptfy.com/a.png?x=1&amp;y=2" property="og:image" />
596            <meta property='og:site_name' content='LMMPTFY'></head><body><meta property="og:title" content="late"></body>"#,
597        )
598        .unwrap();
599        assert_eq!(head.title, "Let Me & MeatProxy");
600        assert_eq!(head.description.as_deref(), Some(r#"plain "desc""#));
601        assert_eq!(head.site_name.as_deref(), Some("LMMPTFY"));
602        assert_eq!(head.image.as_deref(), Some("https://lmmptfy.com/a.png?x=1&y=2"));
603    }
604
605    #[test]
606    fn og_description_beats_the_name_one_and_numeric_entities_decode() {
607        let head = head_of(r#"<head><meta name="description" content="b"><meta property="og:description" content="a &#39;q&#x27; &#128017; &bogus; &amp"><meta property="og:title" content="t"></head>"#).unwrap();
608        assert_eq!(head.description.as_deref(), Some("a 'q' 🐑 &bogus; &amp"));
609    }
610
611    #[test]
612    fn a_missing_og_title_falls_back_to_the_title_element() {
613        let head = head_of("<head><title>\n  Hello\n  there &amp; you </title><meta property=\"og:image\" content=\"/assets/c.png\"></head>").unwrap();
614        assert_eq!(head.title, "Hello there & you");
615        assert_eq!(head.description, None);
616        assert_eq!(head.image.as_deref(), Some("https://lmmptfy.com/assets/c.png"));
617    }
618
619    #[test]
620    fn a_relative_image_is_resolved_against_the_page() {
621        let at = |image: &str| same_site(image, "https://lmmptfy.com/deep/page.html?x#y");
622        assert_eq!(at("/a.png").as_deref(), Some("https://lmmptfy.com/a.png"));
623        assert_eq!(at("a.png").as_deref(), Some("https://lmmptfy.com/deep/a.png"));
624        assert_eq!(at("./a.png").as_deref(), Some("https://lmmptfy.com/deep/a.png"));
625        assert_eq!(at("//lmmptfy.com/a.png").as_deref(), Some("https://lmmptfy.com/a.png"));
626        assert_eq!(at("https://LMMPTFY.com/a.png").as_deref(), Some("https://LMMPTFY.com/a.png"));
627    }
628
629    #[test]
630    fn an_image_off_the_site_is_dropped_and_the_card_kept() {
631        for image in [
632            "https://evil.example/a.png",
633            "http://lmmptfy.com/a.png",
634            "//evil.example/a.png",
635            "https://lmmptfy.com.evil.example/a.png",
636            "https://evil.example/https://lmmptfy.com/a.png",
637            "https://lmmptfy.com@evil.example/a.png",
638            "https://evil.example@lmmptfy.com/a.png",
639            "https://lmmptfy.com:8443/a.png",
640            "https://lmmptfy.com/../a.png",
641            "https://lmmptfy.com\\@evil.example/a.png",
642            "data:image/png;base64,AAAA",
643            "javascript:alert(1)",
644            "ftp://lmmptfy.com/a.png",
645            "",
646        ] {
647            let html = format!(r#"<head><meta property="og:title" content="T"><meta property="og:image" content="{}"></head>"#, image.replace('"', "&quot;"));
648            let head = head_of(&html).unwrap();
649            assert_eq!(head.image, None, "{image}");
650            assert_eq!(head.title, "T");
651        }
652    }
653
654    #[test]
655    fn a_huge_page_is_read_only_to_the_cap_and_text_is_capped() {
656        let wanted = format!(r#"<head><meta property="og:title" content="{}"><meta property="og:description" content="{}">"#, "t".repeat(500), "d".repeat(5_000));
657        let late = format!("{wanted}{}<meta property=\"og:site_name\" content=\"too late\"></head>", " ".repeat(MAX_HTML));
658        let head = head_of(&late).unwrap();
659        assert_eq!(head.site_name, None);
660        assert_eq!(head.title.chars().count(), MAX_TITLE);
661        assert!(head.title.ends_with('…'));
662        assert_eq!(head.description.unwrap().chars().count(), MAX_DESCRIPTION);
663        // A cut inside a multibyte character does not panic.
664        let wide = format!("<head><title>x</title>{}</head>", "é".repeat(MAX_HTML));
665        assert_eq!(head_of(&wide).unwrap().title, "x");
666    }
667
668    #[test]
669    fn comments_scripts_and_the_body_are_not_the_head() {
670        let head = head_of(
671            r#"<head><!-- <meta property="og:title" content="comment"> --><script>var s = '<meta property="og:title" content="script">';</script>
672            <style>/* <title>style</title> */</style><meta property="og:title" content="real"></head>"#,
673        )
674        .unwrap();
675        assert_eq!(head.title, "real");
676    }
677
678    #[test]
679    fn the_route_has_nothing_to_serve_without_a_head_or_an_image() {
680        assert_eq!(card_address(None), None);
681        assert_eq!(card_address(Some(&Head { image: None, ..card() })), None);
682        assert_eq!(card_address(Some(&card())), Some("https://lmmptfy.com/assets/og-card-1200x630-XXvjlRvI.png"));
683    }
684
685    #[test]
686    fn nothing_to_show_is_none() {
687        assert_eq!(head_of(""), None);
688        assert_eq!(head_of("<head><meta charset=utf-8><meta name=viewport content='width=1'></head>"), None);
689        assert_eq!(head_of(r#"<head><meta property="og:image" content="/a.png"><title>   </title></head>"#), None);
690        assert_eq!(head_of("<head><meta property=og:title content="), None);
691        assert_eq!(head_of("<meta <<<>>> =\"= <title"), None);
692    }
693
694    /// Needs the network: the site's real front page, read as the Worker
695    /// would. Run with `cargo test -p lmjtfy --lib -- --ignored real_site --nocapture`.
696    #[test]
697    #[ignore]
698    fn real_site_head() {
699        let out = std::process::Command::new("curl").args(["-sSL", "-A", AGENT, "--max-time", "10", BASE]).output().expect("curl");
700        let html = String::from_utf8_lossy(&out.stdout);
701        let head = extract(&html, BASE).expect("a head");
702        println!("{head:#?}");
703        assert!(!head.title.is_empty());
704        assert!(head.image.as_deref().is_some_and(|image| image.starts_with("https://lmmptfy.com/")));
705    }
706}