1use std::fmt::Write as _;
4
5pub const SANITIZE_TAGS: &[&str] = &[
14 "a",
15 "abbr",
16 "acronym",
17 "address",
18 "b",
19 "big",
20 "blockquote",
21 "br",
22 "cite",
23 "code",
24 "dd",
25 "del",
26 "dfn",
27 "div",
28 "dl",
29 "dt",
30 "em",
31 "figcaption",
32 "figure",
33 "h1",
34 "h2",
35 "h3",
36 "h4",
37 "h5",
38 "h6",
39 "hr",
40 "i",
41 "img",
42 "ins",
43 "kbd",
44 "li",
45 "mark",
46 "ol",
47 "p",
48 "pre",
49 "samp",
50 "small",
51 "span",
52 "strike",
53 "strong",
54 "sub",
55 "sup",
56 "time",
57 "tt",
58 "ul",
59 "var",
60];
61
62pub const SANITIZE_ATTRIBUTES: &[&str] = &[
75 "abbr", "alt", "cite", "class", "datetime", "height", "href", "lang", "name", "src", "title", "width", "xml:lang",
76];
77
78const DROPPED_WITH_CONTENT: &[&str] = &[
80 "script", "style", "template", "noscript", "iframe", "noembed", "noframes", "xmp", "textarea", "title", "svg",
81 "math",
82];
83const VOID: &[&str] = &["br", "hr", "img", "wbr"];
85const URL_ATTRIBUTES: &[&str] = &["href", "src", "cite", "action", "formaction", "poster", "background"];
87const SAFE_SCHEMES: &[&str] = &["http", "https", "mailto", "tel"];
89
90pub fn sanitize(html: &str) -> String {
118 sanitize_with(html, SANITIZE_TAGS, SANITIZE_ATTRIBUTES)
119}
120
121pub fn sanitize_with(html: &str, tags: &[&str], attributes: &[&str]) -> String {
135 Cleaner { tags, attributes, out: String::with_capacity(html.len()), open: Vec::new() }.run(html)
136}
137
138pub fn strip_tags(html: &str) -> String {
153 sanitize_with(html, &[], &[])
154}
155
156pub fn json_escape(json: &str) -> String {
174 let mut out = String::with_capacity(json.len());
175 for c in json.chars() {
176 match c {
177 '<' => out.push_str("\\u003c"),
178 '>' => out.push_str("\\u003e"),
179 '&' => out.push_str("\\u0026"),
180 '\u{2028}' => out.push_str("\\u2028"),
181 '\u{2029}' => out.push_str("\\u2029"),
182 c => out.push(c),
183 }
184 }
185 out
186}
187
188pub fn escape_javascript(text: &str) -> String {
203 let mut out = String::with_capacity(text.len());
204 let mut chars = text.chars().peekable();
205 while let Some(c) = chars.next() {
206 match c {
207 '\\' => out.push_str("\\\\"),
208 '\'' => out.push_str("\\'"),
209 '"' => out.push_str("\\\""),
210 '`' => out.push_str("\\`"),
211 '$' => out.push_str("\\$"),
212 '\r' => {
213 chars.next_if_eq(&'\n');
214 out.push_str("\\n");
215 }
216 '\n' => out.push_str("\\n"),
217 '\u{2028}' => out.push_str("\\u2028"),
218 '\u{2029}' => out.push_str("\\u2029"),
219 '<' if chars.peek() == Some(&'/') => out.push_str("<\\"),
220 c => out.push(c),
221 }
222 }
223 out
224}
225
226struct Cleaner<'a> {
227 tags: &'a [&'a str],
228 attributes: &'a [&'a str],
229 out: String,
230 open: Vec<String>,
232}
233
234impl Cleaner<'_> {
235 fn run(mut self, html: &str) -> String {
236 let mut rest = html;
237 while let Some(start) = rest.find('<') {
238 self.text(&rest[..start]);
239 rest = &rest[start..];
240 rest = self.markup(rest);
241 }
242 self.text(rest);
243 while let Some(name) = self.open.pop() {
244 let _ = write!(self.out, "</{name}>");
245 }
246 self.out
247 }
248
249 fn markup<'h>(&mut self, rest: &'h str) -> &'h str {
251 if let Some(comment) = rest.strip_prefix("<!--") {
252 return comment.find("-->").map_or("", |end| &comment[end + 3..]);
253 }
254 let after = &rest[1..];
255 if after.starts_with(['!', '?']) {
256 return after.find('>').map_or("", |end| &after[end + 1..]);
258 }
259 let closing = after.starts_with('/');
260 let name_start = if closing { &after[1..] } else { after };
261 if !name_start.starts_with(|c: char| c.is_ascii_alphabetic()) {
262 self.out.push_str("<");
263 return after;
264 }
265 let Some((tag, following)) = parse_tag(name_start) else {
266 self.text(rest);
268 return "";
269 };
270 if closing {
271 self.close(&tag.name);
272 return following;
273 }
274 if DROPPED_WITH_CONTENT.contains(&tag.name.as_str()) {
275 return skip_element(following, &tag.name);
276 }
277 if self.tags.contains(&tag.name.as_str()) {
278 self.open_tag(&tag);
279 }
280 following
281 }
282
283 fn open_tag(&mut self, tag: &Tag) {
284 self.out.push('<');
285 self.out.push_str(&tag.name);
286 for (name, value) in &tag.attributes {
287 if !self.attributes.contains(&name.as_str()) || name.starts_with("on") {
288 continue;
289 }
290 if URL_ATTRIBUTES.contains(&name.as_str()) && !safe_url(value) {
291 continue;
292 }
293 let _ = write!(self.out, " {name}=\"");
294 escape_into(&mut self.out, value, true);
295 self.out.push('"');
296 }
297 self.out.push('>');
298 if !VOID.contains(&tag.name.as_str()) {
299 self.open.push(tag.name.clone());
300 }
301 }
302
303 fn close(&mut self, name: &str) {
305 let Some(index) = self.open.iter().rposition(|open| open == name) else { return };
306 for name in self.open.drain(index..).rev() {
307 let _ = write!(self.out, "</{name}>");
308 }
309 }
310
311 fn text(&mut self, text: &str) {
312 escape_into(&mut self.out, text, false);
313 }
314}
315
316struct Tag {
317 name: String,
318 attributes: Vec<(String, String)>,
320}
321
322fn parse_tag(input: &str) -> Option<(Tag, &str)> {
325 let name_end = input.find(|c: char| !(c.is_ascii_alphanumeric() || c == '-' || c == ':')).unwrap_or(input.len());
326 let name = input[..name_end].to_ascii_lowercase();
327 let mut rest = &input[name_end..];
328 let mut attributes = Vec::new();
329 loop {
330 rest = rest.trim_start_matches(|c: char| c.is_ascii_whitespace() || c == '/');
331 if let Some(after) = rest.strip_prefix('>') {
332 return Some((Tag { name, attributes }, after));
333 }
334 if rest.is_empty() {
335 return None;
336 }
337 let attr_end =
338 rest.find(|c: char| c.is_ascii_whitespace() || matches!(c, '=' | '>' | '/')).unwrap_or(rest.len());
339 let attr_end = attr_end.max(rest.chars().next().map_or(1, char::len_utf8));
341 let attr = rest[..attr_end].to_ascii_lowercase();
342 rest = rest[attr_end..].trim_start_matches(|c: char| c.is_ascii_whitespace());
343 let mut value = String::new();
344 if let Some(after) = rest.strip_prefix('=') {
345 let after = after.trim_start_matches(|c: char| c.is_ascii_whitespace());
346 let (raw, following) = match after.chars().next() {
347 Some(quote @ ('"' | '\'')) => {
348 let body = &after[1..];
349 let end = body.find(quote)?;
350 (&body[..end], &body[end + 1..])
351 }
352 _ => {
353 let end = after.find(|c: char| c.is_ascii_whitespace() || c == '>').unwrap_or(after.len());
354 (&after[..end], &after[end..])
355 }
356 };
357 value = decode_entities(raw);
358 rest = following;
359 }
360 attributes.push((attr, value));
361 }
362}
363
364fn skip_element<'h>(input: &'h str, name: &str) -> &'h str {
366 let lower = input.to_ascii_lowercase();
367 let closing = format!("</{name}");
368 let mut from = 0;
369 while let Some(found) = lower[from..].find(&closing) {
370 let at = from + found + closing.len();
371 if lower[at..].starts_with(|c: char| c == '>' || c.is_ascii_whitespace() || c == '/') {
372 return input[at..].find('>').map_or("", |end| &input[at + end + 1..]);
373 }
374 from = at;
375 }
376 ""
377}
378
379fn safe_url(url: &str) -> bool {
381 let url: String =
383 url.chars().filter(|c| !c.is_ascii_control() && *c != ' ').collect::<String>().to_ascii_lowercase();
384 match url.find([':', '/', '?', '#']) {
385 Some(index) if url.as_bytes()[index] == b':' => SAFE_SCHEMES.contains(&&url[..index]),
386 _ => true,
387 }
388}
389
390fn escape_into(out: &mut String, text: &str, attribute: bool) {
392 for (index, c) in text.char_indices() {
393 match c {
394 '<' => out.push_str("<"),
395 '>' => out.push_str(">"),
396 '"' if attribute => out.push_str("""),
397 '&' if attribute || entity_len(&text[index..]).is_none() => out.push_str("&"),
398 c => out.push(c),
399 }
400 }
401}
402
403fn entity_len(text: &str) -> Option<usize> {
405 let body = text.strip_prefix('&')?;
406 let end = body.find(';')?;
407 let name = &body[..end];
408 let valid = if let Some(hex) = name.strip_prefix("#x").or_else(|| name.strip_prefix("#X")) {
409 !hex.is_empty() && hex.len() <= 6 && hex.bytes().all(|b| b.is_ascii_hexdigit())
410 } else if let Some(decimal) = name.strip_prefix('#') {
411 !decimal.is_empty() && decimal.len() <= 7 && decimal.bytes().all(|b| b.is_ascii_digit())
412 } else {
413 (2..=31).contains(&name.len())
414 && name.starts_with(|c: char| c.is_ascii_alphabetic())
415 && name.bytes().all(|b| b.is_ascii_alphanumeric())
416 };
417 valid.then_some(end + 2)
418}
419
420fn decode_entities(text: &str) -> String {
422 let mut out = String::with_capacity(text.len());
423 let mut rest = text;
424 while let Some(start) = rest.find('&') {
425 out.push_str(&rest[..start]);
426 rest = &rest[start..];
427 let body = &rest[1..];
429 let (decoded, used) = if let Some(number) = body.strip_prefix('#') {
430 let (digits, radix) = match number.strip_prefix(['x', 'X']) {
431 Some(hex) => (hex, 16),
432 None => (number, 10),
433 };
434 let len = digits.find(|c: char| !c.is_digit(radix)).unwrap_or(digits.len());
435 let prefix = body.len() - number.len() + (number.len() - digits.len());
436 let value = u32::from_str_radix(&digits[..len], radix).ok().and_then(char::from_u32);
437 let semicolon = usize::from(digits[len..].starts_with(';'));
438 (value, 1 + prefix + len + semicolon)
439 } else {
440 let named = [
441 ("amp;", '&'),
442 ("lt;", '<'),
443 ("gt;", '>'),
444 ("quot;", '"'),
445 ("apos;", '\''),
446 ("colon;", ':'),
447 ("tab;", '\t'),
448 ("newline;", '\n'),
449 ];
450 let lower = body.get(..8).unwrap_or(body).to_ascii_lowercase();
451 named
452 .iter()
453 .find(|(name, _)| lower.starts_with(name))
454 .map_or((None, 1), |(name, c)| (Some(*c), 1 + name.len()))
455 };
456 match decoded {
457 Some(c) => out.push(c),
458 None => out.push('&'),
459 }
460 rest = if decoded.is_some() { &rest[used..] } else { &rest[1..] };
461 }
462 out.push_str(rest);
463 out
464}
465
466#[cfg(test)]
467#[path = "../../tests/security/html.rs"]
468mod tests;