diff --git a/docs/changelog.md b/docs/changelog.md index 661161c6..cc31289d 100644 --- a/docs/changelog.md +++ b/docs/changelog.md @@ -12,6 +12,10 @@ and this project adheres to the [Python Version Specification](https://packaging.python.org/en/latest/specifications/version-specifiers/). See the [Contributing Guide](contributing.md) for details. +## [Unreleased] + +* Update serializer to be non-recursive (#1644). + ## [3.11.0] - 2026-09-25 ### Changed diff --git a/markdown/inlinepatterns.py b/markdown/inlinepatterns.py index 0f3533b2..1908625a 100644 --- a/markdown/inlinepatterns.py +++ b/markdown/inlinepatterns.py @@ -571,10 +571,10 @@ def get_stash(m: re.Match[str]) -> str: id = m.group(1) value = stash.get(id) if value is not None: - try: + if isinstance(value, etree.Element): # Ensure we don't have a placeholder inside a placeholder return self.unescape(self.md.serializer(value)) - except Exception: + else: return r'\%s' % value return util.INLINE_PLACEHOLDER_RE.sub(get_stash, text) diff --git a/markdown/serializers.py b/markdown/serializers.py index 573b2648..255092b4 100644 --- a/markdown/serializers.py +++ b/markdown/serializers.py @@ -48,6 +48,7 @@ from xml.etree.ElementTree import ProcessingInstruction from xml.etree.ElementTree import Comment, ElementTree, Element, QName, HTML_EMPTY import re +from collections import deque from typing import Callable, Literal, NoReturn __all__ = ['to_html_string', 'to_xhtml_string'] @@ -116,60 +117,79 @@ def _escape_attrib_html(text: str) -> str: def _serialize_html(write: Callable[[str], None], elem: Element, format: Literal["html", "xhtml"]) -> None: - tag = elem.tag - text = elem.text - if tag is Comment: - write("" % _escape_cdata(text)) - elif tag is ProcessingInstruction: - write("" % _escape_cdata(text)) - elif tag is None: - if text: - write(_escape_cdata(text)) - for e in elem: - _serialize_html(write, e, format) - else: - namespace_uri = None - if isinstance(tag, QName): - # `QNAME` objects store their data as a string: `{uri}tag` - if tag.text[:1] == "{": - namespace_uri, tag = tag.text[1:].split("}", 1) - else: - raise ValueError('QName objects must define a tag.') - write("<" + tag) - items = elem.items() - if items: - items = sorted(items) # lexical order - for k, v in items: - if isinstance(k, QName): - # Assume a text only `QName` - k = k.text - if isinstance(v, QName): - # Assume a text only `QName` - v = v.text - else: - v = _escape_attrib_html(v) - if k == v and format == 'html': - # handle boolean attributes - write(" %s" % v) - else: - write(' {}="{}"'.format(k, v)) - if namespace_uri: - write(' xmlns="%s"' % (_escape_attrib(namespace_uri))) - if format == "xhtml" and tag.lower() in HTML_EMPTY: - write(" />") - else: - write(">") + stack: deque[Element | str] = deque([elem]) + + count = 0 + while stack: + count += 1 + el = stack.popleft() + + if isinstance(el, str): + write(el) + continue + + tag = el.tag + text = el.text + + if tag is Comment: + write(f"") + elif tag is ProcessingInstruction: + write(f"") + elif tag is None: if text: - if tag.lower() in ["script", "style"]: - write(text) + write(_escape_cdata(text)) + # Add the children in reverse order so we process them in the right order. + if el.tail: + stack.appendleft(_escape_cdata(el.tail)) + stack.extendleft(reversed(el)) + continue + else: + namespace_uri = None + if isinstance(tag, QName): + # `QNAME` objects store their data as a string: `{uri}tag` + if tag.text[:1] == "{": + namespace_uri, tag = tag.text[1:].split("}", 1) else: - write(_escape_cdata(text)) - for e in elem: - _serialize_html(write, e, format) - if tag.lower() not in HTML_EMPTY: - write("") - if elem.tail: - write(_escape_cdata(elem.tail)) + raise ValueError('QName objects must define a tag.') + write(f"<{tag}") + items = el.items() + if items: + items = sorted(items) # lexical order + for k, v in items: + if isinstance(k, QName): + # Assume a text only `QName` + k = k.text + if isinstance(v, QName): + # Assume a text only `QName` + v = v.text + else: + v = _escape_attrib_html(v) + if k == v and format == 'html': + # handle boolean attributes + write(f" {v}") + else: + write(f' {k}="{v}"') + if namespace_uri: + write(f' xmlns="{_escape_attrib(namespace_uri)}"') + if format == "xhtml" and tag.lower() in HTML_EMPTY: + write(" />") + else: + write(">") + if text: + if tag.lower() in ["script", "style"]: + write(text) + else: + write(_escape_cdata(text)) + # Add the tail, end tag, and then the children in reverse order. + # Since we pop from the left, this will ensure results are ordered correctly. + if el.tail: + stack.appendleft(_escape_cdata(el.tail)) + if tag.lower() not in HTML_EMPTY: + stack.appendleft(f"") + stack.extendleft(reversed(el)) + continue + if el.tail: + write(_escape_cdata(el.tail)) def _write_html(root: Element, format: Literal["html", "xhtml"] = "html") -> str: