diff --git a/lib/markdown2.py b/lib/markdown2.py
index d8a00942..756173cb 100755
--- a/lib/markdown2.py
+++ b/lib/markdown2.py
@@ -2371,10 +2371,14 @@ def _unescape_special_chars(self, text: str) -> str:
hashmap = tuple(self._escape_table.items()) + tuple(self._code_table.items())
# html_blocks table is in format {hash: item} compared to usual {item: hash}
hashmap += tuple(tuple(reversed(i)) for i in self.html_blocks.items())
+ replacements = {}
+ for ch, hash in hashmap:
+ replacements.setdefault(hash, ch)
while True:
orig_text = text
- for ch, hash in hashmap:
- text = text.replace(hash, ch)
+ # Scan once per nesting level instead of once for every stored hash.
+ text = re.sub(r'md5-[0-9a-f]{32}',
+ lambda match: replacements.get(match[0], match[0]), text)
if text == orig_text:
break
return text
diff --git a/perf/issue635.py b/perf/issue635.py
new file mode 100644
index 00000000..0cb14c69
--- /dev/null
+++ b/perf/issue635.py
@@ -0,0 +1,47 @@
+"""Reproduce #635 with synthetic distinct code spans; no timing assertions.
+
+Run with the same interpreter before and after the change:
+ python perf/issue635.py
+"""
+import hashlib
+import json
+from pathlib import Path
+import statistics
+import sys
+import time
+
+sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'lib'))
+import markdown2
+
+
+def measure(count):
+ source = '\n\n'.join('`value_%s`' % i for i in range(count))
+ expected = '\n\n'.join('
value_%s
' % i for i in range(count)) + '\n'
+ totals, unescapes = [], []
+ for _ in range(3):
+ converter = markdown2.Markdown()
+ original = converter._unescape_special_chars
+ timings = []
+
+ def timed(text):
+ start = time.perf_counter()
+ result = original(text)
+ timings.append(time.perf_counter() - start)
+ return result
+
+ converter._unescape_special_chars = timed
+ start = time.perf_counter()
+ output = converter.convert(source)
+ totals.append(time.perf_counter() - start)
+ unescapes.append(sum(timings))
+ assert output == expected
+ return {
+ 'spans': count,
+ 'total_seconds_median': statistics.median(totals),
+ 'unescape_seconds_median': statistics.median(unescapes),
+ 'output_sha256': hashlib.sha256(output.encode()).hexdigest(),
+ }
+
+
+if __name__ == '__main__':
+ print(json.dumps({'python': sys.version, 'runs': [measure(n) for n in (500, 1000, 2000, 4000)]}, indent=2))
diff --git a/test/test_markdown2.py b/test/test_markdown2.py
index 0dd22ad9..d7986eed 100755
--- a/test/test_markdown2.py
+++ b/test/test_markdown2.py
@@ -220,6 +220,56 @@ class DirectTestCase(_MarkdownTestCase):
Python-markdown (markdown.py).
"""
+ def test_many_distinct_code_spans(self):
+ source = '\n\n'.join('`value_%s`' % i for i in range(1000))
+ expected = '\n\n'.join('value_%s
' % i for i in range(1000)) + '\n'
+ self.assertEqual(markdown2.markdown(source), expected)
+
+ def test_unescape_nested_tokens(self):
+ md = markdown2.Markdown()
+ md.reset()
+ inner = r'\1\g<0>\\ *'
+ middle = '%s' % markdown2._hash_text(inner)
+ outer = '%s
' % markdown2._hash_text(middle)
+ for entries in ([(inner, markdown2._hash_text(inner)),
+ (middle, markdown2._hash_text(middle))],
+ [(middle, markdown2._hash_text(middle)),
+ (inner, markdown2._hash_text(inner))]):
+ md._code_table = dict(entries)
+ md.html_blocks = {markdown2._hash_text(outer): outer}
+ self.assertEqual(md._unescape_special_chars(markdown2._hash_text(outer)),
+ '%s
' % inner)
+
+ def test_unescape_special_chars_inside_html(self):
+ md = markdown2.Markdown()
+ md.reset()
+ html = '%s %s
' % (md._escape_table['*'], md._escape_table['\\'])
+ token = markdown2._hash_text(html)
+ md.html_blocks[token] = html
+ self.assertEqual(md._unescape_special_chars(token), '* \\
')
+
+ def test_unescape_leaves_unknown_tokens_and_plain_text(self):
+ md = markdown2.Markdown()
+ md.reset()
+ text = 'plain \\ text md5-%s
' % ('0' * 32)
+ self.assertEqual(md._unescape_special_chars(text), text)
+
+ def test_unescape_repeated_tokens(self):
+ md = markdown2.Markdown()
+ md.reset()
+ value = r'\g<0> * literal'
+ token = markdown2._hash_text(value)
+ md._code_table[value] = token
+ self.assertEqual(md._unescape_special_chars('%s %s' % (token, token)), '%s %s' % (value, value))
+
+ def test_unescape_duplicate_token_priority(self):
+ md = markdown2.Markdown()
+ md.reset()
+ token = md._escape_table['*']
+ md._code_table['code value'] = token
+ md.html_blocks[token] = 'HTML value
'
+ self.assertEqual(md._unescape_special_chars(token), '*')
+
def test_slow_hr(self):
import time
text = """\