From 725add3817a6a7cb36a06413191889ad6a4cb05a Mon Sep 17 00:00:00 2001 From: inchang-ing <197932532+inchang-ing@users.noreply.github.com> Date: Wed, 30 Sep 2026 04:07:44 +0800 Subject: [PATCH] fix(parser): fail fast on rules that don't advance state (infinite loop guard) Port markdown-it JS's guard (since 13.0.2): raise RuntimeError when a block/inline rule reports a match without advancing state.line / state.pos, instead of looping forever and exhausting memory. A buggy plugin rule that returns True without advancing made parse() hang indefinitely (e.g. texmath/amsmath plugin bugs, mdit-py-plugins#156 and #117). Built-in rules always advance, so this only affects already broken plugins, which now fail with a clear message. Fixes #441. --- markdown_it/parser_block.py | 12 ++++++++++ markdown_it/parser_inline.py | 22 ++++++++++++++++++ tests/test_rule_guard.py | 44 ++++++++++++++++++++++++++++++++++++ 3 files changed, 78 insertions(+) create mode 100644 tests/test_rule_guard.py diff --git a/markdown_it/parser_block.py b/markdown_it/parser_block.py index 50a7184c..f263100c 100644 --- a/markdown_it/parser_block.py +++ b/markdown_it/parser_block.py @@ -83,9 +83,21 @@ def tokenize(self, state: StateBlock, startLine: int, endLine: int) -> None: # - update `state.line` # - update `state.tokens` # - return True + ok = False for rule in rules: if rule(state, line, endLine, False): + ok = True break + if ok and state.line == line: + # A rule reported a match but did not advance the line. Without + # this guard the parser would loop forever (and usually exhaust + # memory), as markdown-it JS does since 13.0.2. + raise RuntimeError( + f"Block rule {getattr(rule, '__name__', repr(rule))!r} " + "returned True but did not advance state.line, which would " + "cause an infinite loop. Make sure the rule increments " + "state.line when it reports a match." + ) # set state.tight if we had an empty line before current tag # i.e. latest empty line should not count diff --git a/markdown_it/parser_inline.py b/markdown_it/parser_inline.py index af66a7fa..0c482781 100644 --- a/markdown_it/parser_inline.py +++ b/markdown_it/parser_inline.py @@ -143,6 +143,7 @@ def skipToken(self, state: StateInline) -> None: state.pos = cache[pos] return + oldPos = state.pos if state.level < maxNesting: for rule in rules: # Increment state.level and decrement it later to limit recursion. @@ -167,6 +168,16 @@ def skipToken(self, state: StateInline) -> None: # state.pos = state.posMax + if ok and state.pos == oldPos: + # A validation rule reported a match but did not advance the position. + # Without this guard the caller would loop forever, as markdown-it JS + # does since 13.0.2. + raise RuntimeError( + f"Inline rule {getattr(rule, '__name__', repr(rule))!r} " + "returned True but did not advance state.pos, which would cause " + "an infinite loop. Make sure the rule increments state.pos when " + "it reports a match." + ) if not ok: state.pos += 1 cache[pos] = state.pos @@ -186,6 +197,7 @@ def tokenize(self, state: StateInline) -> None: # - update `state.tokens` # - return true + oldPos = state.pos if state.level < maxNesting: for rule in rules: ok = rule(state, False) @@ -195,6 +207,16 @@ def tokenize(self, state: StateInline) -> None: if ok: if state.pos >= end: break + if state.pos == oldPos: + # A rule reported a match but did not advance the position. + # Without this guard the parser would loop forever, as + # markdown-it JS does since 13.0.2. + raise RuntimeError( + f"Inline rule {getattr(rule, '__name__', repr(rule))!r} " + "returned True but did not advance state.pos, which would " + "cause an infinite loop. Make sure the rule increments " + "state.pos when it reports a match." + ) continue state.append_pending(state.src[state.pos]) diff --git a/tests/test_rule_guard.py b/tests/test_rule_guard.py new file mode 100644 index 00000000..d11defa4 --- /dev/null +++ b/tests/test_rule_guard.py @@ -0,0 +1,44 @@ +"""Regression tests for #441: guards against block/inline rules that report a +match without advancing ``state.line`` / ``state.pos`` (which would otherwise +make the parser loop forever, exhausting memory). + +Ported from markdown-it JS, which has thrown on this since 13.0.2. +""" + +import pytest + +from markdown_it import MarkdownIt +from markdown_it.rules_inline.state_inline import StateInline + + +def test_block_rule_guard_raises(): + md = MarkdownIt("commonmark") + # A block rule that claims a match but never advances state.line. + md.block.ruler.before("paragraph", "stuck", lambda state, start, end, silent: True) + with pytest.raises(RuntimeError): + md.parse("text\n") + + +def test_inline_tokenize_guard_raises(): + md = MarkdownIt("commonmark") + # An inline rule that claims a match but never advances state.pos. + md.inline.ruler.before("text", "stuck", lambda state, silent: True) + with pytest.raises(RuntimeError): + md.parse("text\n") + + +def test_inline_skip_token_guard_raises(): + md = MarkdownIt("commonmark") + md.inline.ruler.before("text", "stuck", lambda state, silent: True) + state = StateInline("text", md, {}, []) + with pytest.raises(RuntimeError): + md.inline.skipToken(state) + + +def test_builtin_rules_still_advance(): + # Sanity check: normal parsing is unaffected by the new guards. + md = MarkdownIt("commonmark") + out = md.render("# Hello\n\nA paragraph with *em* and `code`.") + assert "
code" in out