Skip to content
File

Blob: src/client/components/markdownHtmlSanitizer.ts

typescript382 lines
1import { hasChildren, isComment, isDirective, isDocument, isTag, isText } from "domhandler";
2import type { ChildNode, Element, ParentNode } from "domhandler";
3import { parseDocument } from "htmlparser2";
4 
5// This list intentionally mirrors sanitize-html 2.17.1 defaults. Keeping the
6// policy local lets the Worker avoid sanitize-html's Node compatibility needs
7// without quietly widening the Markdown raw-HTML surface.
8const DEFAULT_ALLOWED_TAGS = [
9 "address",
10 "article",
11 "aside",
12 "footer",
13 "header",
14 "h1",
15 "h2",
16 "h3",
17 "h4",
18 "h5",
19 "h6",
20 "hgroup",
21 "main",
22 "nav",
23 "section",
24 "blockquote",
25 "dd",
26 "div",
27 "dl",
28 "dt",
29 "figcaption",
30 "figure",
31 "hr",
32 "li",
33 "menu",
34 "ol",
35 "p",
36 "pre",
37 "ul",
38 "a",
39 "abbr",
40 "b",
41 "bdi",
42 "bdo",
43 "br",
44 "cite",
45 "code",
46 "data",
47 "dfn",
48 "em",
49 "i",
50 "kbd",
51 "mark",
52 "q",
53 "rb",
54 "rp",
55 "rt",
56 "rtc",
57 "ruby",
58 "s",
59 "samp",
60 "small",
61 "span",
62 "strong",
63 "sub",
64 "sup",
65 "time",
66 "u",
67 "var",
68 "wbr",
69 "caption",
70 "col",
71 "colgroup",
72 "table",
73 "tbody",
74 "td",
75 "tfoot",
76 "th",
77 "thead",
78 "tr",
79] satisfies string[];
80 
81const REPO_ALLOWED_TAGS = ["img", "details", "summary", "del", "ins"] satisfies string[];
82const ALLOWED_TAGS = new Set<string>([...DEFAULT_ALLOWED_TAGS, ...REPO_ALLOWED_TAGS]);
83 
84const DROP_WITH_CHILDREN_TAGS = new Set<string>(["script", "style", "textarea", "option"]);
85const ALLOWED_URL_SCHEMES = new Set<string>(["http", "https", "ftp", "mailto", "tel"]);
86const ALLOWED_EMPTY_ATTRIBUTES = new Set<string>(["alt"]);
87const VOID_TAGS = new Set<string>([
88 "area",
89 "base",
90 "basefont",
91 "br",
92 "col",
93 "embed",
94 "hr",
95 "img",
96 "input",
97 "keygen",
98 "link",
99 "meta",
100 "param",
101 "source",
102 "track",
103 "wbr",
104]);
105 
106const ALLOWED_ATTRIBUTES_BY_TAG = new Map<string, ReadonlySet<string>>([
107 ["a", new Set(["href", "title", "name", "target", "rel"])],
108 ["img", new Set(["src", "alt", "title", "loading"])],
109 ["code", new Set(["class"])],
110 ["span", new Set(["class"])],
111 ["pre", new Set(["class"])],
112 ["td", new Set(["align"])],
113 ["th", new Set(["align"])],
114]);
115 
116const ATTRIBUTE_AMPERSAND_RE =
117 /&(?!(?:#[0-9]{1,7}|#x[0-9a-fA-F]{1,6}|[a-zA-Z][a-zA-Z0-9]{1,31});)/g;
118const HTML_CHARACTER_REFERENCE_RE =
119 /&(?:#([0-9]{1,7})|#x([0-9a-fA-F]{1,6})|([a-zA-Z][a-zA-Z0-9]{1,31}));?/g;
120const URL_CONTROL_AND_SPACE_RE = /[\u0000-\u0020]+/g;
121const URL_SCHEME_RE = /^([a-zA-Z][a-zA-Z0-9.+-]*):/;
122 
123const URL_NAMED_CHARACTER_REFERENCES = new Map<string, string>([
124 ["amp", "&"],
125 ["apos", "'"],
126 ["bsol", "\\"],
127 ["colon", ":"],
128 ["gt", ">"],
129 ["lt", "<"],
130 ["NewLine", "\n"],
131 ["quot", '"'],
132 ["sol", "/"],
133 ["Tab", "\t"],
134]);
135 
136function escapeTextValue(value: string): string {
137 return value.replace(/[&<>]/g, (character) => {
138 if (character === "&") {
139 return "&amp;";
140 }
141 if (character === "<") {
142 return "&lt;";
143 }
144 return "&gt;";
145 });
146}
147 
148function escapeAttributeValue(value: string): string {
149 // htmlparser2 decodes character references before this point, so every
150 // attribute value is serialized from a normalized string. That prevents
151 // double-encoding `?a=1&amp;b=2` while still making quotes and brackets inert.
152 return value
153 .replace(ATTRIBUTE_AMPERSAND_RE, "&amp;")
154 .replace(/"/g, "&quot;")
155 .replace(/'/g, "&#39;")
156 .replace(/</g, "&lt;")
157 .replace(/>/g, "&gt;");
158}
159 
160function decodeHtmlCharacterReference(
161 entity: string,
162 decimal: string | undefined,
163 hexadecimal: string | undefined,
164 named: string | undefined
165): string {
166 if (decimal) {
167 return decodeCodePoint(entity, decimal, 10);
168 }
169 
170 if (hexadecimal) {
171 return decodeCodePoint(entity, hexadecimal, 16);
172 }
173 
174 if (named) {
175 return URL_NAMED_CHARACTER_REFERENCES.get(named) ?? entity;
176 }
177 
178 return entity;
179}
180 
181function decodeCodePoint(entity: string, rawCodePoint: string, radix: number): string {
182 const codePoint = Number.parseInt(rawCodePoint, radix);
183 if (!Number.isFinite(codePoint) || codePoint < 0 || codePoint > 0x10ffff) {
184 return entity;
185 }
186 
187 try {
188 return String.fromCodePoint(codePoint);
189 } catch {
190 return entity;
191 }
192}
193 
194function decodeHtmlCharacterReferences(value: string): string {
195 return value.replace(HTML_CHARACTER_REFERENCE_RE, decodeHtmlCharacterReference);
196}
197 
198function removeHtmlCommentsFromUrl(value: string): string {
199 let normalized = value;
200 
201 while (true) {
202 const start = normalized.indexOf("<!--");
203 if (start < 0) {
204 return normalized;
205 }
206 
207 const end = normalized.indexOf("-->", start + 4);
208 if (end < 0) {
209 return normalized;
210 }
211 
212 normalized = `${normalized.slice(0, start)}${normalized.slice(end + 3)}`;
213 }
214}
215 
216function isAllowedUrl(value: string): boolean {
217 // Match sanitize-html's URL-scheme stance: strip characters browsers ignore
218 // in URL protocols, decode references that can hide `javascript:`, then only
219 // reject values with an explicit non-allowed scheme. Relative paths, root
220 // paths, fragments, and protocol-relative URLs therefore remain valid.
221 const decoded = decodeHtmlCharacterReferences(value);
222 const withoutComments = removeHtmlCommentsFromUrl(decoded);
223 const normalized = withoutComments.replace(URL_CONTROL_AND_SPACE_RE, "");
224 const schemeMatch = URL_SCHEME_RE.exec(normalized);
225 if (!schemeMatch) {
226 return true;
227 }
228 
229 return ALLOWED_URL_SCHEMES.has(schemeMatch[1].toLowerCase());
230}
231 
232function isUrlAttribute(attributeName: string): boolean {
233 return attributeName === "href" || attributeName === "src";
234}
235 
236function isAllowedClassName(elementName: string, className: string): boolean {
237 if (elementName === "code") {
238 return className === "hljs" || className.startsWith("language-");
239 }
240 
241 if (elementName === "span") {
242 return className.startsWith("hljs-");
243 }
244 
245 if (elementName === "pre") {
246 return className === "markdown-code-block";
247 }
248 
249 return false;
250}
251 
252function filterClassAttribute(elementName: string, value: string): string {
253 return value
254 .split(/\s+/)
255 .filter((className) => className.length > 0 && isAllowedClassName(elementName, className))
256 .join(" ");
257}
258 
259function sanitizeAttributes(node: Element, elementName: string): void {
260 const allowedAttributes = ALLOWED_ATTRIBUTES_BY_TAG.get(elementName);
261 if (!allowedAttributes) {
262 node.attribs = {};
263 return;
264 }
265 
266 const sanitizedAttributes: Record<string, string> = {};
267 for (const [rawAttributeName, rawValue] of Object.entries(node.attribs)) {
268 const attributeName = rawAttributeName.toLowerCase();
269 if (!allowedAttributes.has(attributeName)) {
270 continue;
271 }
272 
273 const filteredValue =
274 attributeName === "class" ? filterClassAttribute(elementName, rawValue) : rawValue;
275 if (filteredValue.length === 0 && !ALLOWED_EMPTY_ATTRIBUTES.has(attributeName)) {
276 continue;
277 }
278 
279 if (isUrlAttribute(attributeName) && !isAllowedUrl(filteredValue)) {
280 continue;
281 }
282 
283 sanitizedAttributes[attributeName] = filteredValue;
284 }
285 
286 node.attribs = sanitizedAttributes;
287}
288 
289function sanitizeChildren(parent: ParentNode): void {
290 const sanitizedChildren: ChildNode[] = [];
291 for (const child of parent.children) {
292 for (const sanitizedChild of sanitizeNode(child)) {
293 sanitizedChildren.push(sanitizedChild);
294 }
295 }
296 
297 relinkChildren(parent, sanitizedChildren);
298}
299 
300function sanitizeNode(node: ChildNode): ChildNode[] {
301 if (isText(node)) {
302 return [node];
303 }
304 
305 if (isComment(node) || isDirective(node) || isDocument(node)) {
306 return [];
307 }
308 
309 if (!isTag(node)) {
310 if (hasChildren(node)) {
311 sanitizeChildren(node);
312 return node.children;
313 }
314 return [];
315 }
316 
317 const elementName = node.name.toLowerCase();
318 if (DROP_WITH_CHILDREN_TAGS.has(elementName)) {
319 return [];
320 }
321 
322 node.name = elementName;
323 sanitizeChildren(node);
324 
325 if (!ALLOWED_TAGS.has(elementName)) {
326 return node.children;
327 }
328 
329 sanitizeAttributes(node, elementName);
330 return [node];
331}
332 
333function relinkChildren(parent: ParentNode, children: ChildNode[]): void {
334 parent.children = children;
335 
336 children.forEach((child, index) => {
337 child.parent = parent;
338 child.prev = children[index - 1] ?? null;
339 child.next = children[index + 1] ?? null;
340 });
341}
342 
343function serializeAttributes(attributes: Record<string, string>): string {
344 let serialized = "";
345 for (const [name, value] of Object.entries(attributes)) {
346 if (value.length > 0 || ALLOWED_EMPTY_ATTRIBUTES.has(name)) {
347 serialized += ` ${name}="${escapeAttributeValue(value)}"`;
348 } else {
349 serialized += ` ${name}`;
350 }
351 }
352 return serialized;
353}
354 
355function serializeNode(node: ChildNode): string {
356 if (isText(node)) {
357 return escapeTextValue(node.data);
358 }
359 
360 if (!isTag(node)) {
361 return "";
362 }
363 
364 const elementName = node.name.toLowerCase();
365 const attributes = serializeAttributes(node.attribs);
366 if (VOID_TAGS.has(elementName)) {
367 return `<${elementName}${attributes}>`;
368 }
369 
370 return `<${elementName}${attributes}>${serializeChildren(node.children)}</${elementName}>`;
371}
372 
373function serializeChildren(children: readonly ChildNode[]): string {
374 return children.map((child) => serializeNode(child)).join("");
375}
376 
377export function sanitizeMarkdownHtml(rawHtml: string): string {
378 const document = parseDocument(rawHtml, { decodeEntities: true });
379 sanitizeChildren(document);
380 return serializeChildren(document.children);
381}