From 25a32e49a093fc3dd6b8b2d7bfc316e8da3f964a Mon Sep 17 00:00:00 2001 From: CI Test Date: Mon, 5 Oct 2026 22:33:25 +0100 Subject: [PATCH] fix(security): avoid SVG title DOM reparsing Extract strict text-only SVG titles without reparsing model output as DOM, preserving sandboxed SVG rendering and safe accessibility labels. --- static/js/markdown.js | 34 ++++++++++++++++++++------ tests/codeql_security_browser.cjs | 7 +++--- tests/test_markdown_dom_xss_helpers.py | 10 +++++--- 3 files changed, 36 insertions(+), 15 deletions(-) diff --git a/static/js/markdown.js b/static/js/markdown.js index b992c0abb..abf7bcf36 100644 --- a/static/js/markdown.js +++ b/static/js/markdown.js @@ -791,14 +791,32 @@ function renderSvgSandbox(source) { const height = viewBox ? Number(viewBox[2]) : 9; const ratio = Number.isFinite(width / height) && width > 0 && height > 0 ? Math.max(0.5, Math.min(3, width / height)) : (16 / 9); - // XML parsing extracts text without inserting title markup into an HTML DOM. - let title = 'Visual explanation'; - if (typeof DOMParser !== 'undefined') { - const svg = new DOMParser().parseFromString(cleaned, 'image/svg+xml'); - if (!svg.querySelector('parsererror')) { - title = svg.querySelector('svg title')?.textContent?.trim() || title; - } - } + // Extract only a strict text-only SVG title. Do not reparse model output as DOM. + // Nested or malformed title markup falls back to the generic accessible label. + const decodeSvgTitleEntities = value => String(value || '').replace( + /&(?:#([0-9]+)|#x([0-9a-f]+)|(amp|lt|gt|quot|apos));/gi, + (entity, decimal, hex, named) => { + if (named) { + return { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'" }[named.toLowerCase()]; + } + const codePoint = Number.parseInt(decimal || hex, decimal ? 10 : 16); + if ( + !Number.isInteger(codePoint) + || codePoint < 0 + || codePoint > 0x10ffff + || (codePoint >= 0xd800 && codePoint <= 0xdfff) + ) { + return '\uFFFD'; + } + return String.fromCodePoint(codePoint); + }, + ); + + const titleMatch = /]*)?>([^<>]*)<\/title\s*>/i.exec(cleaned); + const extractedTitle = titleMatch + ? decodeSvgTitleEntities(titleMatch[1]).trim() + : ''; + const title = extractedTitle || 'Visual explanation'; const csp = "default-src 'none'; img-src 'none'; media-src 'none'; font-src 'none'; style-src 'unsafe-inline'"; const srcdoc = `${cleaned}`; return `
`; diff --git a/tests/codeql_security_browser.cjs b/tests/codeql_security_browser.cjs index 356a010ba..747957e9f 100644 --- a/tests/codeql_security_browser.cjs +++ b/tests/codeql_security_browser.cjs @@ -61,11 +61,12 @@ const { extractThemeBootstrap } = require('./helpers/theme_bootstrap.cjs'); const valid = addMessage('user', 'In the document, edit this specific text (lines 1–2):\n```\nselected\n```\n\nInstruction: **Keep bold** and `code`'); const titles = [ { source: 'Valid & safe', title: 'Valid & safe' }, - { source: 'Nested bold & text', title: 'Nested bold & text' }, + { source: 'Numeric <safe> & sound', title: 'Numeric & sound' }, + { source: 'Nested bold & text', title: 'Visual explanation' }, { source: '\" onload=\"parent.executed++ <script>', title: '\" onload=\"parent.executed++ nested', title: 'parent.executed++nested' }, - { source: '"><img src=x onerror="parent.executed++">', title: '">' }, + { source: 'nested', title: 'Visual explanation' }, + { source: '"><img src=x onerror="parent.executed++">', title: 'Visual explanation' }, { source: '', title: 'Visual explanation' }, { source: ' ', title: 'Visual explanation' }, { source: 'No title', title: 'Visual explanation' }, diff --git a/tests/test_markdown_dom_xss_helpers.py b/tests/test_markdown_dom_xss_helpers.py index 0c3403706..c00622910 100644 --- a/tests/test_markdown_dom_xss_helpers.py +++ b/tests/test_markdown_dom_xss_helpers.py @@ -7,15 +7,17 @@ from tests.helpers.document_source import document_source _REPO = Path(__file__).resolve().parent.parent -def test_svg_title_extraction_uses_xml_text_without_html_assignment(): +def test_svg_title_extraction_never_reparses_model_output_as_dom(): src = (_REPO / "static" / "js" / "markdown.js").read_text(encoding="utf-8") render = src.split("function renderSvgSandbox(source)", 1)[1].split( "function replaceRawSvgBlocks", 1 )[0] assert "innerHTML" not in render - assert "parseFromString(cleaned, 'image/svg+xml')" in render - assert "textContent" in render - assert "parsererror" in render + assert "DOMParser" not in render + assert "parseFromString" not in render + assert "decodeSvgTitleEntities" in render + assert "([^<>]*)" in render + assert "Visual explanation" in render def test_markdown_raw_html_sanitizer_checks_url_attr_edge_cases():