Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions markdown_it/parser_block.py
Original file line number Diff line number Diff line change
Expand Up @@ -83,9 +83,21 @@ def tokenize(self, state: StateBlock, startLine: int, endLine: int) -> None:
# - update `state.line`
# - update `state.tokens`
# - return True
ok = False
for rule in rules:
if rule(state, line, endLine, False):
ok = True
break
if ok and state.line == line:
# A rule reported a match but did not advance the line. Without
# this guard the parser would loop forever (and usually exhaust
# memory), as markdown-it JS does since 13.0.2.
raise RuntimeError(
f"Block rule {getattr(rule, '__name__', repr(rule))!r} "
"returned True but did not advance state.line, which would "
"cause an infinite loop. Make sure the rule increments "
"state.line when it reports a match."
)

# set state.tight if we had an empty line before current tag
# i.e. latest empty line should not count
Expand Down
22 changes: 22 additions & 0 deletions markdown_it/parser_inline.py
Original file line number Diff line number Diff line change
Expand Up @@ -143,6 +143,7 @@ def skipToken(self, state: StateInline) -> None:
state.pos = cache[pos]
return

oldPos = state.pos
if state.level < maxNesting:
for rule in rules:
# Increment state.level and decrement it later to limit recursion.
Expand All @@ -167,6 +168,16 @@ def skipToken(self, state: StateInline) -> None:
#
state.pos = state.posMax

if ok and state.pos == oldPos:
# A validation rule reported a match but did not advance the position.
# Without this guard the caller would loop forever, as markdown-it JS
# does since 13.0.2.
raise RuntimeError(
f"Inline rule {getattr(rule, '__name__', repr(rule))!r} "
"returned True but did not advance state.pos, which would cause "
"an infinite loop. Make sure the rule increments state.pos when "
"it reports a match."
)
if not ok:
state.pos += 1
cache[pos] = state.pos
Expand All @@ -186,6 +197,7 @@ def tokenize(self, state: StateInline) -> None:
# - update `state.tokens`
# - return true

oldPos = state.pos
if state.level < maxNesting:
for rule in rules:
ok = rule(state, False)
Expand All @@ -195,6 +207,16 @@ def tokenize(self, state: StateInline) -> None:
if ok:
if state.pos >= end:
break
if state.pos == oldPos:
# A rule reported a match but did not advance the position.
# Without this guard the parser would loop forever, as
# markdown-it JS does since 13.0.2.
raise RuntimeError(
f"Inline rule {getattr(rule, '__name__', repr(rule))!r} "
"returned True but did not advance state.pos, which would "
"cause an infinite loop. Make sure the rule increments "
"state.pos when it reports a match."
)
continue

state.append_pending(state.src[state.pos])
Expand Down
44 changes: 44 additions & 0 deletions tests/test_rule_guard.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,44 @@
"""Regression tests for #441: guards against block/inline rules that report a
match without advancing ``state.line`` / ``state.pos`` (which would otherwise
make the parser loop forever, exhausting memory).

Ported from markdown-it JS, which has thrown on this since 13.0.2.
"""

import pytest

from markdown_it import MarkdownIt
from markdown_it.rules_inline.state_inline import StateInline


def test_block_rule_guard_raises():
md = MarkdownIt("commonmark")
# A block rule that claims a match but never advances state.line.
md.block.ruler.before("paragraph", "stuck", lambda state, start, end, silent: True)
with pytest.raises(RuntimeError):
md.parse("text\n")


def test_inline_tokenize_guard_raises():
md = MarkdownIt("commonmark")
# An inline rule that claims a match but never advances state.pos.
md.inline.ruler.before("text", "stuck", lambda state, silent: True)
with pytest.raises(RuntimeError):
md.parse("text\n")


def test_inline_skip_token_guard_raises():
md = MarkdownIt("commonmark")
md.inline.ruler.before("text", "stuck", lambda state, silent: True)
state = StateInline("text", md, {}, [])
with pytest.raises(RuntimeError):
md.inline.skipToken(state)


def test_builtin_rules_still_advance():
# Sanity check: normal parsing is unaffected by the new guards.
md = MarkdownIt("commonmark")
out = md.render("# Hello\n\nA paragraph with *em* and `code`.")
assert "<h1" in out
assert "<em>em</em>" in out
assert "<code>code</code>" in out