Merge PR #126: harden markdown sanitizer with DOMPurify (mXSS) + allowlist/test review fixes

Harden markdown HTML sanitizer with vendored DOMPurify (mXSS)
This commit is contained in:
Ark0N
2026-06-14 22:42:59 +02:00
committed by GitHub
6 changed files with 408 additions and 21 deletions
+2
View File
@@ -67,6 +67,7 @@ appendFileSync(
// 4. Minify frontend assets
run('minify input-cjk.js', 'npx esbuild dist/web/public/input-cjk.js --minify --outfile=dist/web/public/input-cjk.js --allow-overwrite');
run('minify sanitize-html.js', 'npx esbuild dist/web/public/sanitize-html.js --minify --outfile=dist/web/public/sanitize-html.js --allow-overwrite');
run('minify app.js', 'npx esbuild dist/web/public/app.js --minify --outfile=dist/web/public/app.js --allow-overwrite');
run('minify terminal-ui.js', 'npx esbuild dist/web/public/terminal-ui.js --minify --outfile=dist/web/public/terminal-ui.js --allow-overwrite');
run('minify respawn-ui.js', 'npx esbuild dist/web/public/respawn-ui.js --minify --outfile=dist/web/public/respawn-ui.js --allow-overwrite');
@@ -90,6 +91,7 @@ console.log('\n[build] content-hash cache busting');
'notification-manager.js',
'keyboard-accessory.js',
'input-cjk.js',
'sanitize-html.js',
'app.js',
'terminal-ui.js',
'respawn-ui.js',
+7 -21
View File
@@ -1302,28 +1302,14 @@ class CodemanApp {
/** Strip dangerous elements and attributes from HTML (XSS prevention) */
_sanitizeHtml(html) {
const tpl = document.createElement('template');
tpl.innerHTML = html;
const frag = tpl.content;
for (const el of frag.querySelectorAll('script, iframe, object, embed, form, base, meta, link, style')) {
el.remove();
if (typeof window !== 'undefined' && typeof window.sanitizeMarkdownHtml === 'function') {
return window.sanitizeMarkdownHtml(html);
}
for (const el of frag.querySelectorAll('*')) {
for (const attr of [...el.attributes]) {
const name = attr.name.toLowerCase();
if (name.startsWith('on')) {
el.removeAttribute(attr.name);
} else if (['href', 'src', 'action', 'xlink:href', 'formaction'].includes(name)) {
const val = attr.value.replace(/\s/g, '').toLowerCase();
if (val.startsWith('javascript:') || val.startsWith('vbscript:') || val.startsWith('data:text/html')) {
el.removeAttribute(attr.name);
}
}
}
}
const div = document.createElement('div');
div.appendChild(frag);
return div.innerHTML;
// Fail closed: DOMPurify unavailable — never return un-sanitized HTML.
return String(html == null ? '' : html)
.replace(/&/g, '&')
.replace(/</g, '&lt;')
.replace(/>/g, '&gt;');
}
/**
+5
View File
@@ -39,6 +39,9 @@
<script defer src="vendor/xterm-addon-unicode11.min.js"></script>
<script defer src="vendor/xterm-zerolag-input.js"></script>
<script defer src="vendor/marked.min.js"></script>
<!-- DOMPurify (allowlist HTML sanitizer for rendered markdown).
Must load before sanitize-html.js (which wires it) and app.js (which calls it). -->
<script defer src="vendor/dompurify.min.js"></script>
<!-- Synchronous mobile detection — runs before first paint to prevent panel flash -->
<script>if(window.innerWidth<768||(('ontouchstart' in window||navigator.maxTouchPoints>0)&&window.innerWidth<1024))document.documentElement.classList.add('mobile-init');</script>
<!-- Synchronous skin selection — runs before first paint to prevent theme flash -->
@@ -1914,6 +1917,8 @@
<script defer src="notification-manager.js"></script>
<script defer src="keyboard-accessory.js"></script>
<script defer src="input-cjk.js"></script>
<!-- Hardened markdown HTML sanitizer (wires DOMPurify). Must precede app.js. -->
<script defer src="sanitize-html.js"></script>
<script defer src="app.js"></script>
<script defer src="terminal-ui.js"></script>
<script defer src="respawn-ui.js"></script>
+165
View File
@@ -0,0 +1,165 @@
/**
* @fileoverview Allowlist-based HTML sanitizer for markdown-rendered, agent/transcript-derived
* content that is subsequently assigned via innerHTML (response viewer, attachment markdown
* preview, message bodies).
*
* Security (COD-56): the previous sanitizer was a hand-rolled DENYLIST — it removed a fixed
* set of tags (script/iframe/object/embed/form/base/meta/link/style), stripped on* attrs and a
* few dangerous URL schemes, then re-serialized. Denylists are mXSS-prone: they did not strip
* `svg`/`math` (which carry their own foreign-namespace parsing rules and can smuggle script via
* namespace confusion), did not strip `style` attributes (CSS `expression()`/`url(javascript:)`
* on legacy engines), and had no positive allowlist, so any tag/attribute not explicitly named
* survived. `marked` runs with raw-HTML passthrough, so crafted HTML echoed by an agent flows
* straight into this function.
*
* This module replaces that with DOMPurify (Cure53), an allowlist sanitizer that is the
* industry standard for mXSS defense. It is configured to allow exactly the tag/attribute set
* that markdown rendering legitimately produces (headings, lists, code, blockquotes, links,
* tables, images with safe src) and to FORBID `style`/`svg`/`math` plus all event handlers and
* dangerous URL schemes.
*
* Cross-environment: in the browser this file runs as a classic <script> after
* vendor/dompurify.min.js and wires `window.sanitizeMarkdownHtml`. The factory is also exported
* (window/globalThis + CommonJS) so a jsdom unit test can build a sanitizer bound to a
* jsdom-window DOMPurify instance and exercise the exact same config.
*
* @globals {function} sanitizeMarkdownHtml - (html:string) => string, sanitized HTML
* @globals {function} createMarkdownSanitizer - (DOMPurify) => sanitizeMarkdownHtml (for tests)
* @dependency vendor/dompurify.min.js (provides the global DOMPurify)
* @loadorder 5.6 of 15 — after input-cjk.js(5.5), before app.js(6) (app.js calls it)
*/
(function (root) {
'use strict';
/**
* Tags markdown rendering (marked, gfm) legitimately emits. Anything outside this set is
* dropped by DOMPurify. Deliberately excludes svg/math (mXSS foreign-namespace vectors) and
* form/embed/object/iframe/script/style (no place in rendered markdown).
*/
var ALLOWED_TAGS = [
'a',
'b',
'blockquote',
'br',
'caption',
'code',
'del',
'div',
'em',
'h1',
'h2',
'h3',
'h4',
'h5',
'h6',
'hr',
'i',
'img',
'ins',
'kbd',
'li',
'mark',
'ol',
'p',
'pre',
'q',
's',
'samp',
'span',
'strong',
'sub',
'sup',
'table',
'tbody',
'td',
'tfoot',
'th',
'thead',
'tr',
'ul',
'var',
];
/**
* Attributes allowed on the tags above. `style` is intentionally absent (CSS-based vectors).
* `class`/`id` survive because the response viewer adds wrapper classes downstream and code
* blocks may carry `language-*` classes from marked.
*/
var ALLOWED_ATTR = [
'href',
'src',
'alt',
'title',
'class',
'id',
'name',
'colspan',
'rowspan',
'align',
'width',
'height',
'lang',
'dir',
'start',
'reversed',
'type',
];
/**
* Build a sanitizer bound to a specific DOMPurify instance. The browser passes the global
* DOMPurify; tests pass a jsdom-window-bound instance so the same config is exercised under
* vitest without a real browser.
*/
function createMarkdownSanitizer(DOMPurify) {
if (!DOMPurify || typeof DOMPurify.sanitize !== 'function') {
throw new Error('createMarkdownSanitizer: a DOMPurify instance is required');
}
var CONFIG = {
ALLOWED_TAGS: ALLOWED_TAGS,
ALLOWED_ATTR: ALLOWED_ATTR,
// Defense in depth even though style/svg/math are not in ALLOWED_TAGS: also forbid the
// foreign-namespace roots and style so config drift can't silently re-admit them.
FORBID_TAGS: ['style', 'svg', 'math', 'script', 'iframe', 'object', 'embed', 'form'],
FORBID_ATTR: ['style'],
// NOTE: do NOT set USE_PROFILES here. DOMPurify treats USE_PROFILES and
// ALLOWED_TAGS/ALLOWED_ATTR as mutually exclusive — when a profile is set it
// RESETS the allow-lists to the full profile and silently ignores the curated
// lists above, widening the tag set far beyond what markdown emits. Relying on
// the explicit ALLOWED_TAGS/ALLOWED_ATTR keeps the tight allowlist in force;
// FORBID_TAGS/FORBID_ATTR remain as defense-in-depth. DOMPurify still applies
// its default safe-URI handling (blocks javascript:/vbscript:, allows
// http/https/mailto/tel + data: only on image tags).
ALLOW_DATA_ATTR: false,
ADD_ATTR: [],
RETURN_DOM: false,
RETURN_DOM_FRAGMENT: false,
// Keep text content of any removed element (so stripping a stray tag doesn't eat prose),
// matching the previous serializer's behavior of dropping the element but not its text.
KEEP_CONTENT: true,
};
return function sanitizeMarkdownHtml(html) {
return DOMPurify.sanitize(html == null ? '' : String(html), CONFIG);
};
}
// Expose the factory for tests (and any non-browser consumer).
if (root) {
root.createMarkdownSanitizer = createMarkdownSanitizer;
// In the browser, vendor/dompurify.min.js has already defined the global DOMPurify.
if (root.DOMPurify && typeof root.DOMPurify.sanitize === 'function') {
root.sanitizeMarkdownHtml = createMarkdownSanitizer(root.DOMPurify);
}
}
// CommonJS export for the vitest/jsdom unit test.
if (typeof module !== 'undefined' && module.exports) {
module.exports = {
createMarkdownSanitizer: createMarkdownSanitizer,
ALLOWED_TAGS: ALLOWED_TAGS,
ALLOWED_ATTR: ALLOWED_ATTR,
};
}
})(typeof globalThis !== 'undefined' ? globalThis : typeof window !== 'undefined' ? window : this);
File diff suppressed because one or more lines are too long
+226
View File
@@ -0,0 +1,226 @@
/**
* COD-56 — markdown HTML sanitizer (mXSS hardening).
*
* The response viewer / attachment preview render agent- and transcript-derived markdown to
* HTML via `marked` (raw-HTML passthrough) and assign the result with innerHTML. The HTML must
* be sanitized first. The original sanitizer (`_sanitizeHtml` in app.js) was a hand-rolled
* DENYLIST and is mXSS-prone — it never stripped `svg`/`math`/`style`, so foreign-namespace and
* CSS vectors survived.
*
* This suite drives the EXACT shipping artifacts:
* - src/web/public/vendor/dompurify.min.js (the vendored sanitizer)
* - src/web/public/sanitize-html.js (our allowlist config wired to DOMPurify)
*
* It runs in the DEFAULT node environment (it deliberately does NOT declare a per-file jsdom
* environment) and constructs a jsdom window here, then binds the vendored DOMPurify to it. A
* per-file jsdom environment externalizes node:fs/node:path under vite, which made this suite fail
* to load when
* run in isolation (it only survived the full CI run because an earlier node-env test happened to
* pre-cache node:fs). Building the window in-test keeps fs/path native and the suite order-robust.
*
* It feeds a corpus of mXSS payloads (svg/math/style/namespace-confusion/event-handler) and
* asserts the output carries NO script-executing constructs, that the curated allowlist is
* actually enforced (non-markdown tags dropped), and that legitimate markdown-rendered HTML
* survives unchanged. A faithful re-implementation of the OLD denylist is included and asserted to
* LET payloads through — the gap this fix closes.
*
* No port / server needed.
*/
import { describe, it, expect, beforeAll } from 'vitest';
import { readFileSync } from 'node:fs';
import { join } from 'node:path';
import { JSDOM } from 'jsdom';
const publicDir = join(process.cwd(), 'src/web/public');
// One jsdom window shared by the shipping sanitizer (bound to its DOMPurify) and the old-denylist
// reference impl (which needs a DOM `document`).
const dom = new JSDOM('<!DOCTYPE html><html><body></body></html>');
const jsdomWindow = dom.window as unknown as Window & typeof globalThis;
const jsdomDocument = jsdomWindow.document;
/** Build the SHIPPING sanitizer the way the browser does: vendored DOMPurify (bound to our jsdom
* window) + the EXACT CONFIG from sanitize-html.js — so the same allow/forbid lists are exercised
* under vitest without a real browser. */
function loadShippingSanitizer(): (html: string) => string {
const dompurifySrc = readFileSync(join(publicDir, 'vendor/dompurify.min.js'), 'utf8');
const sanitizeSrc = readFileSync(join(publicDir, 'sanitize-html.js'), 'utf8');
// dompurify.min.js is a UMD — evaluate it as CommonJS to obtain the factory (createDOMPurify),
// then bind it to our jsdom window so DOMPurify sanitizes against a real DOM.
const dpModule: { exports: unknown } = { exports: {} };
// eslint-disable-next-line @typescript-eslint/no-implied-eval, no-new-func
new Function('module', 'exports', dompurifySrc)(dpModule, dpModule.exports);
const factory = dpModule.exports as (win: unknown) => { sanitize: (h: string, c?: unknown) => string };
const DOMPurify = factory(jsdomWindow);
// sanitize-html.js exposes createMarkdownSanitizer via its CommonJS export.
const sanModule: { exports: { createMarkdownSanitizer?: (dp: unknown) => (html: string) => string } } = {
exports: {},
};
// eslint-disable-next-line @typescript-eslint/no-implied-eval, no-new-func
new Function('module', 'exports', sanitizeSrc)(sanModule, sanModule.exports);
const create = sanModule.exports.createMarkdownSanitizer;
if (typeof create !== 'function') throw new Error('createMarkdownSanitizer not exported');
const fn = create(DOMPurify);
if (typeof fn !== 'function') throw new Error('sanitizeMarkdownHtml not wired');
return fn;
}
/** Faithful copy of the OLD denylist _sanitizeHtml (app.js pre-COD-56) — used only to prove RED. */
function oldDenylistSanitize(html: string): string {
const tpl = jsdomDocument.createElement('template');
tpl.innerHTML = html;
const frag = tpl.content;
for (const el of frag.querySelectorAll('script, iframe, object, embed, form, base, meta, link, style')) {
el.remove();
}
for (const el of frag.querySelectorAll('*')) {
for (const attr of [...el.attributes]) {
const name = attr.name.toLowerCase();
if (name.startsWith('on')) {
el.removeAttribute(attr.name);
} else if (['href', 'src', 'action', 'xlink:href', 'formaction'].includes(name)) {
const val = attr.value.replace(/\s/g, '').toLowerCase();
if (val.startsWith('javascript:') || val.startsWith('vbscript:') || val.startsWith('data:text/html')) {
el.removeAttribute(attr.name);
}
}
}
}
const div = jsdomDocument.createElement('div');
div.appendChild(frag);
return div.innerHTML;
}
// mXSS / XSS payloads. Each must be neutralized by the shipping sanitizer.
const PAYLOADS: { name: string; html: string }[] = [
{ name: 'img onerror', html: '<img src=x onerror=alert(1)>' },
{ name: 'svg onload', html: '<svg onload=alert(1)></svg>' },
{ name: 'svg/script', html: '<svg><script>alert(1)</script></svg>' },
{ name: 'svg/style mXSS', html: '<svg><style><img src=x onerror=alert(1)></style></svg>' },
{
name: 'math/mtext/table namespace confusion',
html: '<math><mtext><table><mglyph><style><img src=x onerror=alert(1)></style></table></mtext></math>',
},
{ name: 'style attr expression', html: '<div style="width:expression(alert(1))">x</div>' },
{ name: 'style attr url(javascript:)', html: '<div style="background:url(javascript:alert(1))">x</div>' },
{ name: 'style element', html: '<style>body{background:url("javascript:alert(1)")}</style>' },
{ name: 'noscript wrap', html: '<noscript><p title="</noscript><img src=x onerror=alert(1)>">' },
{ name: 'a javascript: href', html: '<a href="javascript:alert(1)">x</a>' },
{ name: 'iframe srcdoc', html: '<iframe srcdoc="<img src=x onerror=alert(1)>"></iframe>' },
{
name: 'foreignObject mXSS',
html: '<svg><foreignObject><iframe src="javascript:alert(1)"></iframe></foreignObject></svg>',
},
{ name: 'details ontoggle', html: '<details open ontoggle=alert(1)>x</details>' },
{ name: 'object data', html: '<object data="javascript:alert(1)"></object>' },
];
function assertNeutralized(out: string, label: string) {
const lower = out.toLowerCase();
expect(lower, `${label}: no <script>`).not.toContain('<script');
expect(lower, `${label}: no <svg>`).not.toContain('<svg');
expect(lower, `${label}: no <math>`).not.toContain('<math');
expect(lower, `${label}: no <iframe>`).not.toContain('<iframe');
expect(lower, `${label}: no <object>`).not.toContain('<object');
expect(lower, `${label}: no <style>`).not.toContain('<style');
expect(lower, `${label}: no onerror`).not.toContain('onerror');
expect(lower, `${label}: no onload`).not.toContain('onload');
expect(lower, `${label}: no ontoggle`).not.toContain('ontoggle');
expect(lower, `${label}: no style= attr`).not.toMatch(/\sstyle\s*=/);
expect(lower, `${label}: no javascript: scheme`).not.toContain('javascript:');
expect(lower, `${label}: no expression(`).not.toContain('expression(');
}
describe('COD-56 markdown sanitizer (DOMPurify allowlist)', () => {
let sanitize: (html: string) => string;
beforeAll(() => {
sanitize = loadShippingSanitizer();
});
describe('mXSS / XSS payloads are neutralized', () => {
for (const { name, html } of PAYLOADS) {
it(`blocks: ${name}`, () => {
assertNeutralized(sanitize(html), name);
});
}
});
describe('curated allowlist is actually enforced (USE_PROFILES must not override it)', () => {
// These tags are in DOMPurify's default html profile but NOT in the curated ALLOWED_TAGS.
// If USE_PROFILES were set, the profile would override the allowlist and these would survive.
const NON_MARKDOWN_TAGS: { name: string; html: string; tag: string }[] = [
{ name: 'button', html: '<button>click</button>', tag: '<button' },
{ name: 'input', html: '<input value="x">', tag: '<input' },
{ name: 'details', html: '<details open>d</details>', tag: '<details' },
{ name: 'audio', html: '<audio controls></audio>', tag: '<audio' },
{ name: 'select/option', html: '<select><option>o</option></select>', tag: '<select' },
{ name: 'label', html: '<label>l</label>', tag: '<label' },
];
for (const { name, html, tag } of NON_MARKDOWN_TAGS) {
it(`drops non-markdown tag: ${name}`, () => {
expect(sanitize(html).toLowerCase()).not.toContain(tag);
});
}
});
describe('legitimate markdown-rendered HTML survives', () => {
it('keeps bold, links, lists, code, headings, tables, safe images', () => {
const md =
'<h2>Title</h2>' +
'<p><strong>bold</strong> and <em>em</em> and <a href="https://example.com">link</a></p>' +
'<ul><li>one</li><li>two</li></ul>' +
'<pre><code class="language-js">const x = 1;</code></pre>' +
'<blockquote><p>quote</p></blockquote>' +
'<table><thead><tr><th>h</th></tr></thead><tbody><tr><td>c</td></tr></tbody></table>' +
'<img src="https://example.com/a.png" alt="pic">';
const out = sanitize(md);
expect(out).toContain('<strong>bold</strong>');
expect(out).toContain('<em>em</em>');
expect(out).toContain('href="https://example.com"');
expect(out).toContain('<li>one</li>');
expect(out).toContain('<code class="language-js">const x = 1;</code>');
expect(out).toContain('<blockquote>');
expect(out).toContain('<th>h</th>');
expect(out).toContain('<td>c</td>');
expect(out).toContain('src="https://example.com/a.png"');
expect(out).toContain('alt="pic"');
});
it('preserves a relative/inline image src and code fences', () => {
const out = sanitize('<p>see <code>code</code></p><img src="/local/path.png" alt="x">');
expect(out).toContain('<code>code</code>');
expect(out).toContain('src="/local/path.png"');
});
});
// RED EVIDENCE: the OLD denylist let mXSS through. This documents the gap the fix closes;
// it asserts the OLD logic FAILS to neutralize at least the svg/math/style vectors.
describe('RED: the old denylist sanitizer was bypassable', () => {
it('old code leaves <svg> / <math> roots in the output', () => {
// svg/math were never in the denylist tag set -> they survive (mXSS foreign namespace).
const svgOut = oldDenylistSanitize('<svg><circle></circle></svg>').toLowerCase();
const mathOut = oldDenylistSanitize('<math><mtext>x</mtext></math>').toLowerCase();
expect(svgOut).toContain('<svg');
expect(mathOut).toContain('<math');
});
it('old code leaves a CSS-vector style attribute in the output', () => {
const out = oldDenylistSanitize('<div style="background:url(javascript:alert(1))">x</div>').toLowerCase();
// style attributes were never stripped by the denylist.
expect(out).toMatch(/\sstyle\s*=/);
expect(out).toContain('javascript:');
});
it('NEW code closes those same gaps', () => {
expect(sanitize('<svg><circle></circle></svg>').toLowerCase()).not.toContain('<svg');
expect(sanitize('<math><mtext>x</mtext></math>').toLowerCase()).not.toContain('<math');
const out = sanitize('<div style="background:url(javascript:alert(1))">x</div>').toLowerCase();
expect(out).not.toMatch(/\sstyle\s*=/);
expect(out).not.toContain('javascript:');
});
});
});