Skip to main content

ocre/
seo.rs

1//! Search engines and language models: JSON-LD, sitemaps and `llms.txt`.
2//!
3//! - [`json_ld`]: structured data (`Organization`, `FAQPage`,
4//!   `BreadcrumbList`...) as a `<script type="application/ld+json">`,
5//!   escaped so the JSON can never close the script element.
6//! - [`Sitemap`]: `/sitemap.xml`, with the other languages of each page
7//!   (`xhtml:link rel="alternate" hreflang`) and `lastmod`.
8//! - [`LlmsTxt`]: `/llms.txt`, the Markdown index of a site for language
9//!   models (<https://llmstxt.org>).
10//!
11//! The canonical and `hreflang` links of a page come from the visitor's
12//! locale: [`I18n::alternate_links`](crate::i18n::I18n::alternate_links).
13//! `ocre g seo` writes `src/seo.rs` with the routes and a list of pages.
14
15use std::fmt::Write as _;
16
17use axum::{
18    http::header,
19    response::{IntoResponse, Response},
20};
21use serde::Serialize;
22
23/// `data` as a JSON-LD `<script>` for a page's `<head>`.
24///
25/// The JSON is escaped for HTML: `<`, `>` and `&` become `\u003c`,
26/// `\u003e` and `\u0026` (as do U+2028 and U+2029), so a value containing
27/// `</script>` cannot end the element, and the JSON parses to the same
28/// data. In askama: `{{ ocre::seo::json_ld(&data)|safe }}`.
29///
30/// # Examples
31///
32/// ```
33/// use serde_json::json;
34///
35/// let html = ocre::seo::json_ld(&json!({
36///     "@context": "https://schema.org",
37///     "@type": "Organization",
38///     "name": "PureFrame </script><b>",
39/// }));
40/// assert!(html.starts_with("<script type=\"application/ld+json\">{"));
41/// assert!(html.contains("PureFrame \\u003c/script\\u003e\\u003cb\\u003e"));
42/// assert_eq!(html.matches("</script>").count(), 1);
43/// ```
44pub fn json_ld(data: &impl Serialize) -> String {
45    let json = serde_json::to_string(data).unwrap_or_else(|_| "null".to_owned());
46    let mut out = String::with_capacity(json.len() + 48);
47    out.push_str("<script type=\"application/ld+json\">");
48    for c in json.chars() {
49        match c {
50            '<' => out.push_str("\\u003c"),
51            '>' => out.push_str("\\u003e"),
52            '&' => out.push_str("\\u0026"),
53            '\u{2028}' => out.push_str("\\u2028"),
54            '\u{2029}' => out.push_str("\\u2029"),
55            c => out.push(c),
56        }
57    }
58    out.push_str("</script>");
59    out
60}
61
62/// A page of a [`Sitemap`]: its absolute URL, last change and other languages.
63#[derive(Debug, Clone, Default, PartialEq, Eq)]
64pub struct SitemapUrl {
65    /// Absolute URL (`https://example.com/fr/pricing`).
66    pub loc: String,
67    /// Last change, `YYYY-MM-DD` or a full W3C date-time; omitted when `None`.
68    pub lastmod: Option<String>,
69    /// The page in each language, the page itself included: `(hreflang, absolute URL)`,
70    /// plus `("x-default", url)` for the language-picking version.
71    pub alternates: Vec<(String, String)>,
72}
73
74/// `/sitemap.xml`: the pages search engines should crawl.
75///
76/// Answer it from a route (it is a response: `application/xml`). One
77/// sitemap holds at most 50,000 URLs (and 50 MB); split larger sites.
78///
79/// # Examples
80///
81/// ```
82/// use ocre::seo::{Sitemap, SitemapUrl};
83///
84/// let mut sitemap = Sitemap::new();
85/// sitemap.add(SitemapUrl {
86///     loc: "https://example.com/en/pricing".into(),
87///     lastmod: Some("2026-10-01".into()),
88///     alternates: vec![
89///         ("en".into(), "https://example.com/en/pricing".into()),
90///         ("fr".into(), "https://example.com/fr/pricing".into()),
91///     ],
92/// });
93/// let xml = sitemap.to_xml();
94/// assert!(xml.contains("<url><loc>https://example.com/en/pricing</loc><lastmod>2026-10-01</lastmod>"));
95/// assert!(xml.contains("<xhtml:link rel=\"alternate\" hreflang=\"fr\" href=\"https://example.com/fr/pricing\"/>"));
96/// ```
97#[derive(Debug, Clone, Default, PartialEq, Eq)]
98pub struct Sitemap {
99    urls: Vec<SitemapUrl>,
100}
101
102impl Sitemap {
103    /// Most URLs in one sitemap file.
104    pub const MAX_URLS: usize = 50_000;
105
106    /// An empty sitemap.
107    pub fn new() -> Self {
108        Self::default()
109    }
110
111    /// Adds a page.
112    pub fn add(&mut self, url: SitemapUrl) -> &mut Self {
113        self.urls.push(url);
114        self
115    }
116
117    /// Adds a page in every language: `urls` is `(hreflang, absolute URL)`
118    /// for each locale, the first being the default (also `x-default`).
119    /// Each version is listed, with all the others as alternates, as
120    /// search engines expect.
121    ///
122    /// # Examples
123    ///
124    /// ```
125    /// let mut sitemap = ocre::seo::Sitemap::new();
126    /// sitemap.add_localized(&[("en", "https://ex.com/en"), ("fr", "https://ex.com/fr")], None);
127    /// let xml = sitemap.to_xml();
128    /// assert_eq!(xml.matches("<url>").count(), 2);
129    /// assert_eq!(xml.matches("hreflang=\"x-default\" href=\"https://ex.com/en\"").count(), 2);
130    /// ```
131    pub fn add_localized(&mut self, urls: &[(&str, &str)], lastmod: Option<&str>) -> &mut Self {
132        let mut alternates: Vec<(String, String)> =
133            urls.iter().map(|(lang, url)| ((*lang).to_owned(), (*url).to_owned())).collect();
134        if let Some((_, default)) = urls.first() {
135            alternates.push(("x-default".to_owned(), (*default).to_owned()));
136        }
137        for (_, url) in urls {
138            self.urls.push(SitemapUrl {
139                loc: (*url).to_owned(),
140                lastmod: lastmod.map(str::to_owned),
141                alternates: alternates.clone(),
142            });
143        }
144        self
145    }
146
147    /// The URLs added so far.
148    pub fn len(&self) -> usize {
149        self.urls.len()
150    }
151
152    /// Whether no URL was added.
153    pub fn is_empty(&self) -> bool {
154        self.urls.is_empty()
155    }
156
157    /// The sitemap document (UTF-8 XML, values escaped).
158    pub fn to_xml(&self) -> String {
159        let mut xml = String::from(
160            "<?xml version=\"1.0\" encoding=\"UTF-8\"?>\n<urlset xmlns=\"http://www.sitemaps.org/schemas/sitemap/0.9\" \
161             xmlns:xhtml=\"http://www.w3.org/1999/xhtml\">\n",
162        );
163        for url in &self.urls {
164            write!(xml, "<url><loc>{}</loc>", escape_xml(&url.loc)).expect("writing to a String");
165            if let Some(lastmod) = &url.lastmod {
166                write!(xml, "<lastmod>{}</lastmod>", escape_xml(lastmod)).expect("writing to a String");
167            }
168            for (lang, href) in &url.alternates {
169                write!(
170                    xml,
171                    "<xhtml:link rel=\"alternate\" hreflang=\"{}\" href=\"{}\"/>",
172                    escape_xml(lang),
173                    escape_xml(href)
174                )
175                .expect("writing to a String");
176            }
177            xml.push_str("</url>\n");
178        }
179        xml.push_str("</urlset>\n");
180        xml
181    }
182}
183
184impl IntoResponse for Sitemap {
185    fn into_response(self) -> Response {
186        ([(header::CONTENT_TYPE, "application/xml; charset=utf-8")], self.to_xml()).into_response()
187    }
188}
189
190/// `/llms.txt`: a site's summary and links in Markdown, for language models (<https://llmstxt.org>).
191///
192/// # Examples
193///
194/// ```
195/// use ocre::seo::LlmsTxt;
196///
197/// let txt = LlmsTxt::new("PureFrame", "AI video upscaling, paid per minute.")
198///     .details("Upload a video, get a free preview, then pay to download the full MP4.")
199///     .link("Pages", "Pricing", "https://pureframe.example/pricing", "per-minute prices")
200///     .link("Pages", "FAQ", "https://pureframe.example/faq", "")
201///     .to_string();
202/// assert!(txt.starts_with("# PureFrame\n\n> AI video upscaling, paid per minute.\n\nUpload a video,"));
203/// let pages = "\n## Pages\n\n- [Pricing](https://pureframe.example/pricing): per-minute prices\n- [FAQ](https://pureframe.example/faq)\n";
204/// assert!(txt.ends_with(pages));
205/// ```
206#[derive(Debug, Clone, Default, PartialEq, Eq)]
207pub struct LlmsTxt {
208    title: String,
209    summary: String,
210    details: String,
211    sections: Vec<(String, Vec<String>)>,
212}
213
214impl LlmsTxt {
215    /// The site's name (`# title`) and one-line summary (`> summary`).
216    pub fn new(title: impl Into<String>, summary: impl Into<String>) -> Self {
217        Self { title: title.into(), summary: summary.into(), ..Self::default() }
218    }
219
220    /// Free text after the summary.
221    #[must_use]
222    pub fn details(mut self, text: impl Into<String>) -> Self {
223        self.details = text.into();
224        self
225    }
226
227    /// A link in the `section` list (sections keep the order they first appear in).
228    #[must_use]
229    pub fn link(mut self, section: &str, title: &str, url: &str, description: &str) -> Self {
230        let line = if description.is_empty() {
231            format!("- [{title}]({url})")
232        } else {
233            format!("- [{title}]({url}): {description}")
234        };
235        match self.sections.iter_mut().find(|(name, _)| name == section) {
236            Some((_, lines)) => lines.push(line),
237            None => self.sections.push((section.to_owned(), vec![line])),
238        }
239        self
240    }
241}
242
243impl std::fmt::Display for LlmsTxt {
244    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
245        write!(f, "# {}\n\n> {}\n", self.title, self.summary)?;
246        if !self.details.is_empty() {
247            write!(f, "\n{}\n", self.details)?;
248        }
249        for (section, lines) in &self.sections {
250            write!(f, "\n## {section}\n\n{}\n", lines.join("\n"))?;
251        }
252        Ok(())
253    }
254}
255
256impl IntoResponse for LlmsTxt {
257    fn into_response(self) -> Response {
258        ([(header::CONTENT_TYPE, "text/plain; charset=utf-8")], self.to_string()).into_response()
259    }
260}
261
262fn escape_xml(text: &str) -> String {
263    let mut out = String::with_capacity(text.len());
264    for c in text.chars() {
265        match c {
266            '&' => out.push_str("&amp;"),
267            '<' => out.push_str("&lt;"),
268            '>' => out.push_str("&gt;"),
269            '"' => out.push_str("&quot;"),
270            '\'' => out.push_str("&apos;"),
271            c => out.push(c),
272        }
273    }
274    out
275}
276
277#[cfg(test)]
278#[path = "../tests/seo.rs"]
279mod tests;