From 08fef493837d383a0490ff9ba90e502cbd31eedc Mon Sep 17 00:00:00 2001 From: wolfgang-aura <169568318+wolfgang-aura@users.noreply.github.com> Date: Wed, 7 Oct 2026 03:27:02 +0800 Subject: [PATCH 1/2] Fix ReDoS from unclosed HTML tags (#707) _sorta_html_tokenize_re could split a run of attributes several ways: the optional namespace group overlapped the attribute name, and the name could start with whitespace that \s+ also matched. An unclosed tag with repeated `a:b=1` or ` a=1` attributes backtracked exponentially. The name now starts at a non-space character and the redundant namespace group is gone, since `:` is already allowed in names. _tag_is_closed counted openers with ``, which rescans the rest of the line from every unclosed `` only up to the end of the current line, which matches what the regex counted. Add the three shapes from the issue to test/test_redos.py. Fixes #707 --- CHANGES.md | 1 + lib/markdown2.py | 22 +++++++++++++++++++--- test/test_redos.py | 18 ++++++++++++++++++ 3 files changed, 38 insertions(+), 3 deletions(-) diff --git a/CHANGES.md b/CHANGES.md index 8911b5cb..fc2170ce 100644 --- a/CHANGES.md +++ b/CHANGES.md @@ -2,6 +2,7 @@ ## python-markdown2 2.5.6 (not yet released) +- [pull #NNN] Fix excessive CPU use in the inline HTML tokenizer on repeated unclosed tag fragments (#707) - [pull #730] Fix `tables` extra splitting multi-backtick code spans at a leading pipe and merging cells after a code span containing literal backticks. - [pull #729] Fix `tables` extra merging cells when a pipe directly follows a code span, as in compact rows like `|`-v`|verbose|`. - [pull #725] Fix `tables` extra dropping escaped pipes at the end of header and body rows. diff --git a/lib/markdown2.py b/lib/markdown2.py index b3c29847..fe726298 100755 --- a/lib/markdown2.py +++ b/lib/markdown2.py @@ -1171,7 +1171,24 @@ def _tag_is_closed(self, tag_name: str, text: str) -> bool: return True # check if number of open tags == number of close tags - if len(re.findall('<%s(?:.*?)>' % tag_name, text)) != text.count('' % tag_name): + # (a linear scan: `` re-scans the rest of the line from every + # unclosed ` line_end: + line_end = text.find('\n', pos) + if line_end == -1: + line_end = len(text) + end = text.find('>', pos, line_end) + if end == -1: + pos = text.find(open_tag, line_end + 1) + else: + open_count += 1 + pos = text.find(open_tag, end + 1) + if open_count != text.count('' % tag_name): return False # check that close tag position is AFTER open tag @@ -1349,8 +1366,7 @@ def _run_span_gamut(self, text: str) -> str: (?:\w+) # tag name (?: # attributes \s+ # whitespace after tag - (?:[^\t<>"'=/]+:)? - [^<>"'=/]+= # attr name + [^\s<>"'=/][^<>"'=/]*= # attr name, can't start with whitespace (?:"[^"]*?"|'[^']*?'|[^<>"'=/\s]+) # value, quoted or unquoted. If unquoted, no spaces allowed )* \s*/?> diff --git a/test/test_redos.py b/test/test_redos.py index 3bea176f..2951ac2d 100644 --- a/test/test_redos.py +++ b/test/test_redos.py @@ -43,6 +43,21 @@ def issue_668(): return 'a_b **x***y* c_d' +def issue_707_unclosed_tags(): + # https://github.com/trentm/python-markdown2/issues/707 + return '

Date: Wed, 7 Oct 2026 09:38:07 +0800 Subject: [PATCH 2/2] Add PR number to changelog entry --- CHANGES.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGES.md b/CHANGES.md index fc2170ce..2077c78b 100644 --- a/CHANGES.md +++ b/CHANGES.md @@ -2,7 +2,7 @@ ## python-markdown2 2.5.6 (not yet released) -- [pull #NNN] Fix excessive CPU use in the inline HTML tokenizer on repeated unclosed tag fragments (#707) +- [pull #732] Fix excessive CPU use in the inline HTML tokenizer on repeated unclosed tag fragments (#707) - [pull #730] Fix `tables` extra splitting multi-backtick code spans at a leading pipe and merging cells after a code span containing literal backticks. - [pull #729] Fix `tables` extra merging cells when a pipe directly follows a code span, as in compact rows like `|`-v`|verbose|`. - [pull #725] Fix `tables` extra dropping escaped pipes at the end of header and body rows.