ocre/seo.rs
1//! Search engines and language models: JSON-LD, sitemaps and `llms.txt`.
2//!
3//! - [`json_ld`]: structured data (`Organization`, `FAQPage`,
4//! `BreadcrumbList`...) as a `<script type="application/ld+json">`,
5//! escaped so the JSON can never close the script element.
6//! - [`Sitemap`]: `/sitemap.xml`, with the other languages of each page
7//! (`xhtml:link rel="alternate" hreflang`) and `lastmod`.
8//! - [`LlmsTxt`]: `/llms.txt`, the Markdown index of a site for language
9//! models (<https://llmstxt.org>).
10//!
11//! The canonical and `hreflang` links of a page come from the visitor's
12//! locale: [`I18n::alternate_links`](crate::i18n::I18n::alternate_links).
13//! `ocre g seo` writes `src/seo.rs` with the routes and a list of pages.
14
15use std::fmt::Write as _;
16
17use axum::{
18 http::header,
19 response::{IntoResponse, Response},
20};
21use serde::Serialize;
22
23/// `data` as a JSON-LD `<script>` for a page's `<head>`.
24///
25/// The JSON is escaped for HTML: `<`, `>` and `&` become `\u003c`,
26/// `\u003e` and `\u0026` (as do U+2028 and U+2029), so a value containing
27/// `</script>` cannot end the element, and the JSON parses to the same
28/// data. In askama: `{{ ocre::seo::json_ld(&data)|safe }}`.
29///
30/// # Examples
31///
32/// ```
33/// use serde_json::json;
34///
35/// let html = ocre::seo::json_ld(&json!({
36/// "@context": "https://schema.org",
37/// "@type": "Organization",
38/// "name": "PureFrame </script><b>",
39/// }));
40/// assert!(html.starts_with("<script type=\"application/ld+json\">{"));
41/// assert!(html.contains("PureFrame \\u003c/script\\u003e\\u003cb\\u003e"));
42/// assert_eq!(html.matches("</script>").count(), 1);
43/// ```
44pub fn json_ld(data: &impl Serialize) -> String {
45 let json = serde_json::to_string(data).unwrap_or_else(|_| "null".to_owned());
46 let mut out = String::with_capacity(json.len() + 48);
47 out.push_str("<script type=\"application/ld+json\">");
48 for c in json.chars() {
49 match c {
50 '<' => out.push_str("\\u003c"),
51 '>' => out.push_str("\\u003e"),
52 '&' => out.push_str("\\u0026"),
53 '\u{2028}' => out.push_str("\\u2028"),
54 '\u{2029}' => out.push_str("\\u2029"),
55 c => out.push(c),
56 }
57 }
58 out.push_str("</script>");
59 out
60}
61
62/// A page of a [`Sitemap`]: its absolute URL, last change and other languages.
63#[derive(Debug, Clone, Default, PartialEq, Eq)]
64pub struct SitemapUrl {
65 /// Absolute URL (`https://example.com/fr/pricing`).
66 pub loc: String,
67 /// Last change, `YYYY-MM-DD` or a full W3C date-time; omitted when `None`.
68 pub lastmod: Option<String>,
69 /// The page in each language, the page itself included: `(hreflang, absolute URL)`,
70 /// plus `("x-default", url)` for the language-picking version.
71 pub alternates: Vec<(String, String)>,
72}
73
74/// `/sitemap.xml`: the pages search engines should crawl.
75///
76/// Answer it from a route (it is a response: `application/xml`). One
77/// sitemap holds at most 50,000 URLs (and 50 MB); split larger sites.
78///
79/// # Examples
80///
81/// ```
82/// use ocre::seo::{Sitemap, SitemapUrl};
83///
84/// let mut sitemap = Sitemap::new();
85/// sitemap.add(SitemapUrl {
86/// loc: "https://example.com/en/pricing".into(),
87/// lastmod: Some("2026-10-01".into()),
88/// alternates: vec![
89/// ("en".into(), "https://example.com/en/pricing".into()),
90/// ("fr".into(), "https://example.com/fr/pricing".into()),
91/// ],
92/// });
93/// let xml = sitemap.to_xml();
94/// assert!(xml.contains("<url><loc>https://example.com/en/pricing</loc><lastmod>2026-10-01</lastmod>"));
95/// assert!(xml.contains("<xhtml:link rel=\"alternate\" hreflang=\"fr\" href=\"https://example.com/fr/pricing\"/>"));
96/// ```
97#[derive(Debug, Clone, Default, PartialEq, Eq)]
98pub struct Sitemap {
99 urls: Vec<SitemapUrl>,
100}
101
102impl Sitemap {
103 /// Most URLs in one sitemap file.
104 pub const MAX_URLS: usize = 50_000;
105
106 /// An empty sitemap.
107 pub fn new() -> Self {
108 Self::default()
109 }
110
111 /// Adds a page.
112 pub fn add(&mut self, url: SitemapUrl) -> &mut Self {
113 self.urls.push(url);
114 self
115 }
116
117 /// Adds a page in every language: `urls` is `(hreflang, absolute URL)`
118 /// for each locale, the first being the default (also `x-default`).
119 /// Each version is listed, with all the others as alternates, as
120 /// search engines expect.
121 ///
122 /// # Examples
123 ///
124 /// ```
125 /// let mut sitemap = ocre::seo::Sitemap::new();
126 /// sitemap.add_localized(&[("en", "https://ex.com/en"), ("fr", "https://ex.com/fr")], None);
127 /// let xml = sitemap.to_xml();
128 /// assert_eq!(xml.matches("<url>").count(), 2);
129 /// assert_eq!(xml.matches("hreflang=\"x-default\" href=\"https://ex.com/en\"").count(), 2);
130 /// ```
131 pub fn add_localized(&mut self, urls: &[(&str, &str)], lastmod: Option<&str>) -> &mut Self {
132 let mut alternates: Vec<(String, String)> =
133 urls.iter().map(|(lang, url)| ((*lang).to_owned(), (*url).to_owned())).collect();
134 if let Some((_, default)) = urls.first() {
135 alternates.push(("x-default".to_owned(), (*default).to_owned()));
136 }
137 for (_, url) in urls {
138 self.urls.push(SitemapUrl {
139 loc: (*url).to_owned(),
140 lastmod: lastmod.map(str::to_owned),
141 alternates: alternates.clone(),
142 });
143 }
144 self
145 }
146
147 /// The URLs added so far.
148 pub fn len(&self) -> usize {
149 self.urls.len()
150 }
151
152 /// Whether no URL was added.
153 pub fn is_empty(&self) -> bool {
154 self.urls.is_empty()
155 }
156
157 /// The sitemap document (UTF-8 XML, values escaped).
158 pub fn to_xml(&self) -> String {
159 let mut xml = String::from(
160 "<?xml version=\"1.0\" encoding=\"UTF-8\"?>\n<urlset xmlns=\"http://www.sitemaps.org/schemas/sitemap/0.9\" \
161 xmlns:xhtml=\"http://www.w3.org/1999/xhtml\">\n",
162 );
163 for url in &self.urls {
164 write!(xml, "<url><loc>{}</loc>", escape_xml(&url.loc)).expect("writing to a String");
165 if let Some(lastmod) = &url.lastmod {
166 write!(xml, "<lastmod>{}</lastmod>", escape_xml(lastmod)).expect("writing to a String");
167 }
168 for (lang, href) in &url.alternates {
169 write!(
170 xml,
171 "<xhtml:link rel=\"alternate\" hreflang=\"{}\" href=\"{}\"/>",
172 escape_xml(lang),
173 escape_xml(href)
174 )
175 .expect("writing to a String");
176 }
177 xml.push_str("</url>\n");
178 }
179 xml.push_str("</urlset>\n");
180 xml
181 }
182}
183
184impl IntoResponse for Sitemap {
185 fn into_response(self) -> Response {
186 ([(header::CONTENT_TYPE, "application/xml; charset=utf-8")], self.to_xml()).into_response()
187 }
188}
189
190/// `/llms.txt`: a site's summary and links in Markdown, for language models (<https://llmstxt.org>).
191///
192/// # Examples
193///
194/// ```
195/// use ocre::seo::LlmsTxt;
196///
197/// let txt = LlmsTxt::new("PureFrame", "AI video upscaling, paid per minute.")
198/// .details("Upload a video, get a free preview, then pay to download the full MP4.")
199/// .link("Pages", "Pricing", "https://pureframe.example/pricing", "per-minute prices")
200/// .link("Pages", "FAQ", "https://pureframe.example/faq", "")
201/// .to_string();
202/// assert!(txt.starts_with("# PureFrame\n\n> AI video upscaling, paid per minute.\n\nUpload a video,"));
203/// let pages = "\n## Pages\n\n- [Pricing](https://pureframe.example/pricing): per-minute prices\n- [FAQ](https://pureframe.example/faq)\n";
204/// assert!(txt.ends_with(pages));
205/// ```
206#[derive(Debug, Clone, Default, PartialEq, Eq)]
207pub struct LlmsTxt {
208 title: String,
209 summary: String,
210 details: String,
211 sections: Vec<(String, Vec<String>)>,
212}
213
214impl LlmsTxt {
215 /// The site's name (`# title`) and one-line summary (`> summary`).
216 pub fn new(title: impl Into<String>, summary: impl Into<String>) -> Self {
217 Self { title: title.into(), summary: summary.into(), ..Self::default() }
218 }
219
220 /// Free text after the summary.
221 #[must_use]
222 pub fn details(mut self, text: impl Into<String>) -> Self {
223 self.details = text.into();
224 self
225 }
226
227 /// A link in the `section` list (sections keep the order they first appear in).
228 #[must_use]
229 pub fn link(mut self, section: &str, title: &str, url: &str, description: &str) -> Self {
230 let line = if description.is_empty() {
231 format!("- [{title}]({url})")
232 } else {
233 format!("- [{title}]({url}): {description}")
234 };
235 match self.sections.iter_mut().find(|(name, _)| name == section) {
236 Some((_, lines)) => lines.push(line),
237 None => self.sections.push((section.to_owned(), vec![line])),
238 }
239 self
240 }
241}
242
243impl std::fmt::Display for LlmsTxt {
244 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
245 write!(f, "# {}\n\n> {}\n", self.title, self.summary)?;
246 if !self.details.is_empty() {
247 write!(f, "\n{}\n", self.details)?;
248 }
249 for (section, lines) in &self.sections {
250 write!(f, "\n## {section}\n\n{}\n", lines.join("\n"))?;
251 }
252 Ok(())
253 }
254}
255
256impl IntoResponse for LlmsTxt {
257 fn into_response(self) -> Response {
258 ([(header::CONTENT_TYPE, "text/plain; charset=utf-8")], self.to_string()).into_response()
259 }
260}
261
262fn escape_xml(text: &str) -> String {
263 let mut out = String::with_capacity(text.len());
264 for c in text.chars() {
265 match c {
266 '&' => out.push_str("&"),
267 '<' => out.push_str("<"),
268 '>' => out.push_str(">"),
269 '"' => out.push_str("""),
270 '\'' => out.push_str("'"),
271 c => out.push(c),
272 }
273 }
274 out
275}
276
277#[cfg(test)]
278#[path = "../tests/seo.rs"]
279mod tests;