Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
188 changes: 185 additions & 3 deletions lib/NodeUtils.js
Original file line number Diff line number Diff line change
Expand Up @@ -121,6 +121,16 @@ function escapeAttr(s) {
});
}

function serializeForeignRawText(element) {
var s = '';
for (var child = element.firstChild; child; child = child.nextSibling) {
s += child.nodeType === 3 /*TEXT_NODE*/ ||
child.nodeType === 4 /*CDATA_SECTION_NODE*/ ?
escape(child.data) : serializeOne(child, element);
}
return s;
}

function attrname(a) {
var ns = a.namespaceURI;
if (!ns)
Expand Down Expand Up @@ -289,6 +299,169 @@ function escapeProcessingInstructionContent(rawContent) {
: rawContent;
}

var foreignContextCache = new WeakMap();

// Namespaces a re-parsing HTML parser would assign, derived from the ancestor
// chain. A parser reading our output only sees tag names, so `<math><desc>`
// puts `desc` in the MathML namespace while `<svg><desc>` makes it an HTML
// integration point. Deciding by element name alone gets those two cases wrong.
var SVG_INTEGRATION_POINTS = {
foreignobject: true,
desc: true,
title: true
};

var MATHML_TEXT_INTEGRATION_POINTS = {
mi: true,
mo: true,
mn: true,
ms: true,
mtext: true
};

// Start tags that a parser treats as a parse error inside foreign content: it
// pops back out to HTML instead of nesting them. Everything below such a tag is
// therefore parsed as HTML, whatever the SVG/MathML ancestors say.
// https://html.spec.whatwg.org/multipage/parsing.html#parsing-main-inforeign
var HTML_BREAKOUT_TAGS = {
b: true, big: true, blockquote: true, body: true, br: true, center: true,
code: true, dd: true, div: true, dl: true, dt: true, em: true, embed: true,
h1: true, h2: true, h3: true, h4: true, h5: true, h6: true, head: true,
hr: true, i: true, img: true, li: true, listing: true, menu: true,
meta: true, nobr: true, ol: true, p: true, pre: true, ruby: true, s: true,
small: true, span: true, strong: true, strike: true, sub: true, sup: true,
table: true, tt: true, u: true, ul: true, var: true
};

// `font` only breaks out when it carries one of these attributes.
function isBreakoutFont(element) {
for (var i = 0; i < element._numattrs; i++) {
var name = utils.toASCIILowerCase(attrname(element._attr(i)));
if (name === 'color' || name === 'face' || name === 'size') return true;
}
return false;
}

function breaksOutOfForeignContent(element, name) {
return HTML_BREAKOUT_TAGS[name] === true ||
(name === 'font' && isBreakoutFont(element));
}

function isHtmlAnnotationXml(element) {
for (var i = 0; i < element._numattrs; i++) {
var attribute = element._attr(i);
if (utils.toASCIILowerCase(attrname(attribute)) === 'encoding') {
var encoding = attribute.value && utils.toASCIILowerCase(attribute.value);
return encoding === 'text/html' || encoding === 'application/xhtml+xml';
}
}
return false;
}

// The namespace that `childName` is parsed in when it appears inside `element`,
// given that `element` itself is parsed in `parentNamespace`.
function childNamespace(element, parentNamespace, childName) {
var name = utils.toASCIILowerCase(serializedTagName(element) || '');

if (parentNamespace === NAMESPACE.SVG) {
// Integration points are per-namespace: `desc`/`title` open an HTML island
// inside SVG only.
if (SVG_INTEGRATION_POINTS[name] || breaksOutOfForeignContent(element, name))
return NAMESPACE.HTML;
return NAMESPACE.SVG;
}

if (parentNamespace === NAMESPACE.MATHML) {
if (MATHML_TEXT_INTEGRATION_POINTS[name]) {
// `mglyph` and `malignmark` stay in MathML even inside a text
// integration point.
return (childName === 'mglyph' || childName === 'malignmark') ?
NAMESPACE.MATHML : NAMESPACE.HTML;
}
if (name === 'annotation-xml')
return isHtmlAnnotationXml(element) ? NAMESPACE.HTML : NAMESPACE.MATHML;
return breaksOutOfForeignContent(element, name) ?
NAMESPACE.HTML : NAMESPACE.MATHML;
}

if (name === 'svg') return NAMESPACE.SVG;
if (name === 'math') return NAMESPACE.MATHML;
return NAMESPACE.HTML;
}

// Resolves the namespace a re-parsing parser would put `kid` in. Ancestors are
// collected upwards only until a cached one is found, then resolved downwards,
// so repeated serialization of siblings costs one step instead of a full walk.
// The cache stores the namespace an element itself is parsed in, which -- unlike
// the namespace of its children -- does not depend on which child is serialized.
function serializationNamespace(parent, kid) {
if (parent.nodeType === 0 && kid.parentNode)
parent = kid.parentNode;

var clock = parent.rooted && parent.ownerDocument.modclock;
var chain = [];
var ns = null;

for (var node = parent; node;) {
if (node.nodeType === 1 /*ELEMENT_NODE*/) {
var cached = clock && foreignContextCache.get(node);
chain.push(node);
if (cached && cached.document === node.ownerDocument &&
cached.clock === clock) {
// The cached value is the namespace of this element itself, so the
// downward pass still has to run its own step.
ns = cached.namespace;
break;
}
node = node.parentNode;
} else if (node.nodeType === 11 /*DOCUMENT_FRAGMENT_NODE*/ && node._host) {
node = node._host;
} else {
node = node.parentNode;
}
}

if (chain.length === 0) return ns === null ? NAMESPACE.HTML : ns;

if (ns === null) {
// No cached ancestor: a detached subtree is serialized on its own, so its
// root carries the namespace a parser would have inferred from ancestors
// we cannot see.
var root = chain[chain.length - 1];
ns = (root.namespaceURI === NAMESPACE.SVG ||
root.namespaceURI === NAMESPACE.MATHML) ? root.namespaceURI :
NAMESPACE.HTML;
}

for (var index = chain.length - 1, cacheable = !!clock; index >= 0; index--) {
var element = chain[index];
if (cacheable) {
foreignContextCache.set(element, {
clock: clock,
document: element.ownerDocument,
namespace: ns
});
}
// Below an `annotation-xml` the namespace depends on its `encoding`
// attribute, and attribute edits do not bump the document's mod clock, so
// those descendants must not be cached.
if (ns === NAMESPACE.MATHML &&
utils.toASCIILowerCase(serializedTagName(element) || '') ===
'annotation-xml')
cacheable = false;
var childName = index > 0 ?
utils.toASCIILowerCase(serializedTagName(chain[index - 1]) || '') :
utils.toASCIILowerCase(serializedTagName(kid) || '');
ns = childNamespace(element, ns, childName);
}

return ns;
}

function isInForeignContent(parent, kid) {
return serializationNamespace(parent, kid) !== NAMESPACE.HTML;
}

function serializeOne(kid, parent) {
var s = '';
switch(kid.nodeType) {
Expand All @@ -307,11 +480,20 @@ function serializeOne(kid, parent) {
s += '>';

if (!(html && emptyElements[tagname])) {
var ss = kid.serialize();
var upperTag = tagname.toUpperCase();
// If an element can have raw content, this content may
// potentially require escaping to avoid XSS.
var upperTag = tagname.toUpperCase();
if (hasRawContent[upperTag] && !hasRawContentFallback[upperTag] && ss.includes('</')) {
var nonFallbackRawContent = hasRawContent[upperTag] &&
!hasRawContentFallback[upperTag];
var escapeRawText = html && nonFallbackRawContent && !kid._innerHTML &&
isInForeignContent(parent, kid);
var ss = escapeRawText ? serializeForeignRawText(kid) : kid.serialize();
// Escape the element's own closing tag even when we believe we are in
// foreign content: a breakout tag earlier in the same foreign container
// pops the parser back to HTML, and then this element *is* raw text for
// it. Escaping here costs nothing when we were right -- in foreign
// content a literal `</tag>` can only come from already-escaped data.
if (nonFallbackRawContent && ss.includes('</')) {
ss = escapeMatchingClosingTag(ss, tagname);
const fallbackTags = fallbackRawContentTags(parent);
for (const fallbackTag of fallbackTags) {
Expand Down
Loading
Loading