Skip to main content

ocre/security/
html.rs

1//! Cleaning user-supplied HTML and escaping values for scripts.
2
3use std::fmt::Write as _;
4
5/// Tags [`sanitize`] keeps: Rails' safe list (`Rails::HTML5::SafeListSanitizer`).
6///
7/// # Examples
8///
9/// ```
10/// assert!(ocre::security::SANITIZE_TAGS.contains(&"strong"));
11/// assert!(!ocre::security::SANITIZE_TAGS.contains(&"script"));
12/// ```
13pub const SANITIZE_TAGS: &[&str] = &[
14    "a",
15    "abbr",
16    "acronym",
17    "address",
18    "b",
19    "big",
20    "blockquote",
21    "br",
22    "cite",
23    "code",
24    "dd",
25    "del",
26    "dfn",
27    "div",
28    "dl",
29    "dt",
30    "em",
31    "figcaption",
32    "figure",
33    "h1",
34    "h2",
35    "h3",
36    "h4",
37    "h5",
38    "h6",
39    "hr",
40    "i",
41    "img",
42    "ins",
43    "kbd",
44    "li",
45    "mark",
46    "ol",
47    "p",
48    "pre",
49    "samp",
50    "small",
51    "span",
52    "strike",
53    "strong",
54    "sub",
55    "sup",
56    "time",
57    "tt",
58    "ul",
59    "var",
60];
61
62/// Attributes [`sanitize`] keeps on those tags: Rails' safe list.
63///
64/// `href`, `src` and `cite` also need an `http:`, `https:`, `mailto:` or
65/// `tel:` URL (or a relative one). `style` and event handlers (`onclick`...)
66/// are never kept.
67///
68/// # Examples
69///
70/// ```
71/// assert!(ocre::security::SANITIZE_ATTRIBUTES.contains(&"href"));
72/// assert!(!ocre::security::SANITIZE_ATTRIBUTES.contains(&"style"));
73/// ```
74pub const SANITIZE_ATTRIBUTES: &[&str] = &[
75    "abbr", "alt", "cite", "class", "datetime", "height", "href", "lang", "name", "src", "title", "width", "xml:lang",
76];
77
78/// Elements removed with their content, whatever the allow list says.
79const DROPPED_WITH_CONTENT: &[&str] = &[
80    "script", "style", "template", "noscript", "iframe", "noembed", "noframes", "xmp", "textarea", "title", "svg",
81    "math",
82];
83/// Elements without a closing tag.
84const VOID: &[&str] = &["br", "hr", "img", "wbr"];
85/// Attributes holding a URL, whose scheme is checked.
86const URL_ATTRIBUTES: &[&str] = &["href", "src", "cite", "action", "formaction", "poster", "background"];
87/// URL schemes kept in URL attributes.
88const SAFE_SCHEMES: &[&str] = &["http", "https", "mailto", "tel"];
89
90/// Cleans user-supplied HTML down to [`SANITIZE_TAGS`] and [`SANITIZE_ATTRIBUTES`] (Rails' `sanitize`).
91///
92/// The output is rebuilt, not filtered: kept tags are written back with
93/// quoted, escaped attributes; other tags are removed but their text kept;
94/// `<script>`, `<style>`, `<iframe>`, `<svg>` and similar are removed with
95/// their content; comments are removed; `href`/`src` with a scheme other
96/// than `http`, `https`, `mailto` or `tel` (e.g. `javascript:`) are removed;
97/// text is escaped; unclosed tags are closed. The result is safe to insert
98/// with askama's `|safe`: `{{ comment.body_html|safe }}`.
99///
100/// Sanitize when saving (store the clean HTML) rather than on every render:
101/// it costs CPU proportional to the input, about 1 ms per 100 KB.
102///
103/// # Examples
104///
105/// ```
106/// use ocre::security::sanitize;
107///
108/// assert_eq!(
109///     sanitize(r#"<p onclick="steal()">Hi <b>there</b><script>alert(1)</script></p>"#),
110///     "<p>Hi <b>there</b></p>"
111/// );
112/// assert_eq!(sanitize(r#"<a href="javascript:alert(1)">x</a>"#), "<a>x</a>");
113/// assert_eq!(sanitize(r#"<a href="https://example.com" target="_blank">x</a>"#), r#"<a href="https://example.com">x</a>"#);
114/// assert_eq!(sanitize("<em>unclosed"), "<em>unclosed</em>");
115/// assert_eq!(sanitize("1 < 2 & 3"), "1 &lt; 2 &amp; 3");
116/// ```
117pub fn sanitize(html: &str) -> String {
118    sanitize_with(html, SANITIZE_TAGS, SANITIZE_ATTRIBUTES)
119}
120
121/// Like [`sanitize`], with your own allowed tags and attributes (Rails' `sanitize(html, tags:, attributes:)`).
122///
123/// Names are lowercase. Script-like elements, event handlers and unsafe URL
124/// schemes stay out whatever the lists say.
125///
126/// # Examples
127///
128/// ```
129/// use ocre::security::sanitize_with;
130///
131/// let html = r#"<p class="x"><a href="/about" title="About">About</a> <img src="x.png"></p>"#;
132/// assert_eq!(sanitize_with(html, &["a"], &["href"]), r#"<a href="/about">About</a> "#);
133/// ```
134pub fn sanitize_with(html: &str, tags: &[&str], attributes: &[&str]) -> String {
135    Cleaner { tags, attributes, out: String::with_capacity(html.len()), open: Vec::new() }.run(html)
136}
137
138/// Removes every tag and comment and keeps the text, escaped (Rails' `strip_tags`).
139///
140/// `<script>` and `<style>` content is removed too. The result is HTML
141/// (`&lt;` stays escaped): insert it with `|safe`, or decode it for plain
142/// text.
143///
144/// # Examples
145///
146/// ```
147/// use ocre::security::strip_tags;
148///
149/// assert_eq!(strip_tags("<p>Hello <b>world</b>!</p><!-- note -->"), "Hello world!");
150/// assert_eq!(strip_tags("<script>alert(1)</script>a &lt; b"), "a &lt; b");
151/// ```
152pub fn strip_tags(html: &str) -> String {
153    sanitize_with(html, &[], &[])
154}
155
156/// Escapes a JSON string for a `<script>` element (Rails' `json_escape`).
157///
158/// `<`, `>` and `&` become `\u003c`, `\u003e` and `\u0026`, and U+2028 and
159/// U+2029 become escapes, so the JSON cannot close the script element or
160/// break JavaScript parsing; it still parses to the same value.
161///
162/// # Examples
163///
164/// ```
165/// use ocre::security::json_escape;
166///
167/// let json = serde_json::json!({ "title": "</script><script>alert(1)</script>" }).to_string();
168/// let safe = json_escape(&json);
169/// assert_eq!(safe, r#"{"title":"\u003c/script\u003e\u003cscript\u003ealert(1)\u003c/script\u003e"}"#);
170/// assert_eq!(serde_json::from_str::<serde_json::Value>(&safe)?["title"], "</script><script>alert(1)</script>");
171/// # Ok::<(), serde_json::Error>(())
172/// ```
173pub fn json_escape(json: &str) -> String {
174    let mut out = String::with_capacity(json.len());
175    for c in json.chars() {
176        match c {
177            '<' => out.push_str("\\u003c"),
178            '>' => out.push_str("\\u003e"),
179            '&' => out.push_str("\\u0026"),
180            '\u{2028}' => out.push_str("\\u2028"),
181            '\u{2029}' => out.push_str("\\u2029"),
182            c => out.push(c),
183        }
184    }
185    out
186}
187
188/// Escapes text for a JavaScript string literal in single, double or back quotes (Rails' `escape_javascript`).
189///
190/// Backslashes, quotes, `` ` ``, `$`, newlines, U+2028/U+2029 and `</` are
191/// escaped. Prefer passing data as JSON ([`json_escape`]) or `data-`
192/// attributes; with a Content-Security-Policy, inline scripts also need a
193/// nonce ([`CspNonce`](super::CspNonce)).
194///
195/// # Examples
196///
197/// ```
198/// use ocre::security::escape_javascript;
199///
200/// assert_eq!(escape_javascript("It's \"fine\"\n</script>"), r#"It\'s \"fine\"\n<\/script>"#);
201/// ```
202pub fn escape_javascript(text: &str) -> String {
203    let mut out = String::with_capacity(text.len());
204    let mut chars = text.chars().peekable();
205    while let Some(c) = chars.next() {
206        match c {
207            '\\' => out.push_str("\\\\"),
208            '\'' => out.push_str("\\'"),
209            '"' => out.push_str("\\\""),
210            '`' => out.push_str("\\`"),
211            '$' => out.push_str("\\$"),
212            '\r' => {
213                chars.next_if_eq(&'\n');
214                out.push_str("\\n");
215            }
216            '\n' => out.push_str("\\n"),
217            '\u{2028}' => out.push_str("\\u2028"),
218            '\u{2029}' => out.push_str("\\u2029"),
219            '<' if chars.peek() == Some(&'/') => out.push_str("<\\"),
220            c => out.push(c),
221        }
222    }
223    out
224}
225
226struct Cleaner<'a> {
227    tags: &'a [&'a str],
228    attributes: &'a [&'a str],
229    out: String,
230    /// Kept elements not closed yet, innermost last.
231    open: Vec<String>,
232}
233
234impl Cleaner<'_> {
235    fn run(mut self, html: &str) -> String {
236        let mut rest = html;
237        while let Some(start) = rest.find('<') {
238            self.text(&rest[..start]);
239            rest = &rest[start..];
240            rest = self.markup(rest);
241        }
242        self.text(rest);
243        while let Some(name) = self.open.pop() {
244            let _ = write!(self.out, "</{name}>");
245        }
246        self.out
247    }
248
249    /// Handles the markup at the start of `rest` (which starts with `<`) and returns what follows.
250    fn markup<'h>(&mut self, rest: &'h str) -> &'h str {
251        if let Some(comment) = rest.strip_prefix("<!--") {
252            return comment.find("-->").map_or("", |end| &comment[end + 3..]);
253        }
254        let after = &rest[1..];
255        if after.starts_with(['!', '?']) {
256            // Doctype, CDATA, processing instruction: removed.
257            return after.find('>').map_or("", |end| &after[end + 1..]);
258        }
259        let closing = after.starts_with('/');
260        let name_start = if closing { &after[1..] } else { after };
261        if !name_start.starts_with(|c: char| c.is_ascii_alphabetic()) {
262            self.out.push_str("&lt;");
263            return after;
264        }
265        let Some((tag, following)) = parse_tag(name_start) else {
266            // Never closed: the rest is text.
267            self.text(rest);
268            return "";
269        };
270        if closing {
271            self.close(&tag.name);
272            return following;
273        }
274        if DROPPED_WITH_CONTENT.contains(&tag.name.as_str()) {
275            return skip_element(following, &tag.name);
276        }
277        if self.tags.contains(&tag.name.as_str()) {
278            self.open_tag(&tag);
279        }
280        following
281    }
282
283    fn open_tag(&mut self, tag: &Tag) {
284        self.out.push('<');
285        self.out.push_str(&tag.name);
286        for (name, value) in &tag.attributes {
287            if !self.attributes.contains(&name.as_str()) || name.starts_with("on") {
288                continue;
289            }
290            if URL_ATTRIBUTES.contains(&name.as_str()) && !safe_url(value) {
291                continue;
292            }
293            let _ = write!(self.out, " {name}=\"");
294            escape_into(&mut self.out, value, true);
295            self.out.push('"');
296        }
297        self.out.push('>');
298        if !VOID.contains(&tag.name.as_str()) {
299            self.open.push(tag.name.clone());
300        }
301    }
302
303    /// Closes `name` and the kept elements opened inside it; ignores it when it is not open.
304    fn close(&mut self, name: &str) {
305        let Some(index) = self.open.iter().rposition(|open| open == name) else { return };
306        for name in self.open.drain(index..).rev() {
307            let _ = write!(self.out, "</{name}>");
308        }
309    }
310
311    fn text(&mut self, text: &str) {
312        escape_into(&mut self.out, text, false);
313    }
314}
315
316struct Tag {
317    name: String,
318    /// Lowercase names, entity-decoded values.
319    attributes: Vec<(String, String)>,
320}
321
322/// Parses `name attr="value" ...>` and returns the tag and what follows `>`,
323/// or `None` when the tag never ends.
324fn parse_tag(input: &str) -> Option<(Tag, &str)> {
325    let name_end = input.find(|c: char| !(c.is_ascii_alphanumeric() || c == '-' || c == ':')).unwrap_or(input.len());
326    let name = input[..name_end].to_ascii_lowercase();
327    let mut rest = &input[name_end..];
328    let mut attributes = Vec::new();
329    loop {
330        rest = rest.trim_start_matches(|c: char| c.is_ascii_whitespace() || c == '/');
331        if let Some(after) = rest.strip_prefix('>') {
332            return Some((Tag { name, attributes }, after));
333        }
334        if rest.is_empty() {
335            return None;
336        }
337        let attr_end =
338            rest.find(|c: char| c.is_ascii_whitespace() || matches!(c, '=' | '>' | '/')).unwrap_or(rest.len());
339        // A lone `=` or other junk still moves forward by at least one character.
340        let attr_end = attr_end.max(rest.chars().next().map_or(1, char::len_utf8));
341        let attr = rest[..attr_end].to_ascii_lowercase();
342        rest = rest[attr_end..].trim_start_matches(|c: char| c.is_ascii_whitespace());
343        let mut value = String::new();
344        if let Some(after) = rest.strip_prefix('=') {
345            let after = after.trim_start_matches(|c: char| c.is_ascii_whitespace());
346            let (raw, following) = match after.chars().next() {
347                Some(quote @ ('"' | '\'')) => {
348                    let body = &after[1..];
349                    let end = body.find(quote)?;
350                    (&body[..end], &body[end + 1..])
351                }
352                _ => {
353                    let end = after.find(|c: char| c.is_ascii_whitespace() || c == '>').unwrap_or(after.len());
354                    (&after[..end], &after[end..])
355                }
356            };
357            value = decode_entities(raw);
358            rest = following;
359        }
360        attributes.push((attr, value));
361    }
362}
363
364/// Skips to after `</name>` (any case), or to the end.
365fn skip_element<'h>(input: &'h str, name: &str) -> &'h str {
366    let lower = input.to_ascii_lowercase();
367    let closing = format!("</{name}");
368    let mut from = 0;
369    while let Some(found) = lower[from..].find(&closing) {
370        let at = from + found + closing.len();
371        if lower[at..].starts_with(|c: char| c == '>' || c.is_ascii_whitespace() || c == '/') {
372            return input[at..].find('>').map_or("", |end| &input[at + end + 1..]);
373        }
374        from = at;
375    }
376    ""
377}
378
379/// Whether a (decoded) URL is relative or uses a safe scheme.
380fn safe_url(url: &str) -> bool {
381    // Browsers ignore tabs and newlines in URLs and trim control characters and spaces.
382    let url: String =
383        url.chars().filter(|c| !c.is_ascii_control() && *c != ' ').collect::<String>().to_ascii_lowercase();
384    match url.find([':', '/', '?', '#']) {
385        Some(index) if url.as_bytes()[index] == b':' => SAFE_SCHEMES.contains(&&url[..index]),
386        _ => true,
387    }
388}
389
390/// Escapes `<`, `>`, `"` (in attributes) and `&`, keeping well-formed entity references in text.
391fn escape_into(out: &mut String, text: &str, attribute: bool) {
392    for (index, c) in text.char_indices() {
393        match c {
394            '<' => out.push_str("&lt;"),
395            '>' => out.push_str("&gt;"),
396            '"' if attribute => out.push_str("&quot;"),
397            '&' if attribute || entity_len(&text[index..]).is_none() => out.push_str("&amp;"),
398            c => out.push(c),
399        }
400    }
401}
402
403/// Length of the entity reference at the start of `text` (`&amp;`, `&#39;`, `&#x27;`), if it is one.
404fn entity_len(text: &str) -> Option<usize> {
405    let body = text.strip_prefix('&')?;
406    let end = body.find(';')?;
407    let name = &body[..end];
408    let valid = if let Some(hex) = name.strip_prefix("#x").or_else(|| name.strip_prefix("#X")) {
409        !hex.is_empty() && hex.len() <= 6 && hex.bytes().all(|b| b.is_ascii_hexdigit())
410    } else if let Some(decimal) = name.strip_prefix('#') {
411        !decimal.is_empty() && decimal.len() <= 7 && decimal.bytes().all(|b| b.is_ascii_digit())
412    } else {
413        (2..=31).contains(&name.len())
414            && name.starts_with(|c: char| c.is_ascii_alphabetic())
415            && name.bytes().all(|b| b.is_ascii_alphanumeric())
416    };
417    valid.then_some(end + 2)
418}
419
420/// Decodes numeric references and the common named ones; others stay as written.
421fn decode_entities(text: &str) -> String {
422    let mut out = String::with_capacity(text.len());
423    let mut rest = text;
424    while let Some(start) = rest.find('&') {
425        out.push_str(&rest[..start]);
426        rest = &rest[start..];
427        // Browsers accept numeric references without the `;` in attributes.
428        let body = &rest[1..];
429        let (decoded, used) = if let Some(number) = body.strip_prefix('#') {
430            let (digits, radix) = match number.strip_prefix(['x', 'X']) {
431                Some(hex) => (hex, 16),
432                None => (number, 10),
433            };
434            let len = digits.find(|c: char| !c.is_digit(radix)).unwrap_or(digits.len());
435            let prefix = body.len() - number.len() + (number.len() - digits.len());
436            let value = u32::from_str_radix(&digits[..len], radix).ok().and_then(char::from_u32);
437            let semicolon = usize::from(digits[len..].starts_with(';'));
438            (value, 1 + prefix + len + semicolon)
439        } else {
440            let named = [
441                ("amp;", '&'),
442                ("lt;", '<'),
443                ("gt;", '>'),
444                ("quot;", '"'),
445                ("apos;", '\''),
446                ("colon;", ':'),
447                ("tab;", '\t'),
448                ("newline;", '\n'),
449            ];
450            let lower = body.get(..8).unwrap_or(body).to_ascii_lowercase();
451            named
452                .iter()
453                .find(|(name, _)| lower.starts_with(name))
454                .map_or((None, 1), |(name, c)| (Some(*c), 1 + name.len()))
455        };
456        match decoded {
457            Some(c) => out.push(c),
458            None => out.push('&'),
459        }
460        rest = if decoded.is_some() { &rest[used..] } else { &rest[1..] };
461    }
462    out.push_str(rest);
463    out
464}
465
466#[cfg(test)]
467#[path = "../../tests/security/html.rs"]
468mod tests;