diff --git a/myst_parser/mdit_to_docutils/base.py b/myst_parser/mdit_to_docutils/base.py index 68cfae72..0d853b6f 100644 --- a/myst_parser/mdit_to_docutils/base.py +++ b/myst_parser/mdit_to_docutils/base.py @@ -482,10 +482,11 @@ def renderInlineAsText(self, tokens: list[SyntaxTreeNode]) -> str: # noqa: N802 result = "" for token in tokens or []: - if token.type == "text": + if token.type in {"text", "text_special"}: + # An escape or entity is a text_special token with no children. + # Its content is the character the alt text must keep + # (executablebooks/MyST-Parser#1210). result += token.content - # elif token.type == "image": - # result += self.renderInlineAsText(token.children) else: result += self.renderInlineAsText(token.children or []) return result diff --git a/tests/test_image_alt.py b/tests/test_image_alt.py new file mode 100644 index 00000000..3f13405b --- /dev/null +++ b/tests/test_image_alt.py @@ -0,0 +1,30 @@ +"""Image alt text keeps CommonMark escapes and entities.""" + +from docutils.core import publish_parts + +from myst_parser.parsers.docutils_ import Parser + + +def _html(source: str) -> str: + return publish_parts(source, parser=Parser(), writer_name="html5")["body"].strip() + + +def test_image_alt_keeps_escapes_and_entities(): + """Escapes and entities inside an image alt are not dropped. + + markdown-it-py represents them as ``text_special`` tokens with no children. + Reading only ``text`` tokens used to drop the character + (executablebooks/MyST-Parser#1210). The same source in a link already + rendered the character. + """ + assert _html(r"") == '


