From 29a30201fd6be2344f9f036f3651fed7ddd10fb1 Mon Sep 17 00:00:00 2001 From: Gyanu Mayank Date: Tue, 6 Oct 2026 22:34:32 +0530 Subject: [PATCH] =?UTF-8?q?=F0=9F=90=9B=20FIX:=20Keep=20backslash=20escape?= =?UTF-8?q?s=20and=20entities=20in=20image=20alt=20text?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit text_join only converted the text_special tokens that are direct children of an inline token. An image keeps its alt text as its own child token list, so escapes and entities there were never turned into text and the alt renderer dropped them: ![a \* b](/u) gave alt="a b". Join the children of each child token too, as markdown-it 15.0.0 does. Fixes #445 --- CHANGELOG.md | 1 + markdown_it/rules_core/text_join.py | 64 +++++++++++++------------ tests/test_port/fixtures/issue-fixes.md | 19 ++++++++ 3 files changed, 54 insertions(+), 30 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 040ff138..0697a7c5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,7 @@ * ✨ Add `--enable-tables` to the CLI for file, standard input and interactive parsing in [#422](https://github.com/executablebooks/markdown-it-py/pull/422) * šŸ› Fix CLI interactive mode joining input lines with an extra newline, which split every line into its own paragraph and broke hard line breaks, in [#172](https://github.com/executablebooks/markdown-it-py/issues/172) * šŸ› Fix trimming and splitting with the Python whitespace set instead of the CommonMark one, which dropped U+001C–U+001F and U+0085 from paragraphs, headings, table cells and fence info strings and let distinct reference labels resolve each other, in [#418](https://github.com/executablebooks/markdown-it-py/pull/418), thanks to [@Nexory](https://github.com/Nexory) +* šŸ› Fix image `alt` text dropping backslash escapes and entities, by also joining `text_special` tokens inside an image (ported from markdown-it 15.0.0), in [#445](https://github.com/executablebooks/markdown-it-py/issues/445) * šŸ“š Document the Python renderer constructor contract. * šŸ”§ Drop the optional dependency on the `commonmark` package and its benchmark, since the upstream package is unmaintained, in [#401](https://github.com/executablebooks/markdown-it-py/pull/401) diff --git a/markdown_it/rules_core/text_join.py b/markdown_it/rules_core/text_join.py index 939b83b2..73c36ec5 100644 --- a/markdown_it/rules_core/text_join.py +++ b/markdown_it/rules_core/text_join.py @@ -19,35 +19,39 @@ def text_join(state: StateCore) -> None: if inline_token.type != "inline": continue - # convert text_special to text and join all adjacent text nodes - new_tokens: list[Token] = [] children = inline_token.children or [] - i = 0 - while i < len(children): - child_token = children[i] - if child_token.type == "text_special": - child_token.type = "text" - if ( - child_token.type == "text" - and new_tokens - and new_tokens[-1].type == "text" - ): - # Collapse a run of adjacent text nodes in a single join, instead - # of pairwise `a + b` concatenation. The pairwise form is O(L*k) - # in the size of the run because each step rebuilds the growing - # prefix; "".join is O(L). - parts = [new_tokens[-1].content, child_token.content] + for child_token in children: + # image `alt` is parsed into its own token tree + if child_token.children: + child_token.children = _join_text_tokens(child_token.children) + inline_token.children = _join_text_tokens(children) + + +def _join_text_tokens(tokens: list[Token]) -> list[Token]: + """Convert `text_special` to `text` and join all adjacent text tokens""" + new_tokens: list[Token] = [] + i = 0 + while i < len(tokens): + token = tokens[i] + if token.type == "text_special": + token.type = "text" + if token.type == "text" and new_tokens and new_tokens[-1].type == "text": + # Collapse a run of adjacent text nodes in a single join, instead + # of pairwise `a + b` concatenation. The pairwise form is O(L*k) + # in the size of the run because each step rebuilds the growing + # prefix; "".join is O(L). + parts = [new_tokens[-1].content, token.content] + i += 1 + while i < len(tokens): + next_token = tokens[i] + if next_token.type == "text_special": + next_token.type = "text" + if next_token.type != "text": + break + parts.append(next_token.content) i += 1 - while i < len(children): - next_token = children[i] - if next_token.type == "text_special": - next_token.type = "text" - if next_token.type != "text": - break - parts.append(next_token.content) - i += 1 - new_tokens[-1].content = "".join(parts) - else: - new_tokens.append(child_token) - i += 1 - inline_token.children = new_tokens + new_tokens[-1].content = "".join(parts) + else: + new_tokens.append(token) + i += 1 + return new_tokens diff --git a/tests/test_port/fixtures/issue-fixes.md b/tests/test_port/fixtures/issue-fixes.md index b630fcee..ef1d9310 100644 --- a/tests/test_port/fixtures/issue-fixes.md +++ b/tests/test_port/fixtures/issue-fixes.md @@ -54,3 +54,22 @@ Fix parsing of incorrect numeric character references

&#X22y; &#35y;

. + +#445 Image alt keeps backslash escapes and entities +. +![a \* b](/u) + +![a & b](/u) + +![a < b](/u) + +![a \* *b* & c](/u) + +![C:\Python26](/u) +. +

a * b

+

a & b

+

a < b

+

a * b & c

+

C:\Python26

+.