Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 6 additions & 2 deletions lib/markdown2.py
Original file line number Diff line number Diff line change
Expand Up @@ -2371,10 +2371,14 @@ def _unescape_special_chars(self, text: str) -> str:
hashmap = tuple(self._escape_table.items()) + tuple(self._code_table.items())
# html_blocks table is in format {hash: item} compared to usual {item: hash}
hashmap += tuple(tuple(reversed(i)) for i in self.html_blocks.items())
replacements = {}
for ch, hash in hashmap:
replacements.setdefault(hash, ch)
while True:
orig_text = text
for ch, hash in hashmap:
text = text.replace(hash, ch)
# Scan once per nesting level instead of once for every stored hash.
text = re.sub(r'md5-[0-9a-f]{32}',
lambda match: replacements.get(match[0], match[0]), text)
if text == orig_text:
break
return text
Expand Down
47 changes: 47 additions & 0 deletions perf/issue635.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
"""Reproduce #635 with synthetic distinct code spans; no timing assertions.

Run with the same interpreter before and after the change:
python perf/issue635.py
"""
import hashlib
import json
from pathlib import Path
import statistics
import sys
import time

sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'lib'))
import markdown2


def measure(count):
source = '\n\n'.join('`value_%s`' % i for i in range(count))
expected = '\n\n'.join('<p><code>value_%s</code></p>' % i for i in range(count)) + '\n'
totals, unescapes = [], []
for _ in range(3):
converter = markdown2.Markdown()
original = converter._unescape_special_chars
timings = []

def timed(text):
start = time.perf_counter()
result = original(text)
timings.append(time.perf_counter() - start)
return result

converter._unescape_special_chars = timed
start = time.perf_counter()
output = converter.convert(source)
totals.append(time.perf_counter() - start)
unescapes.append(sum(timings))
assert output == expected
return {
'spans': count,
'total_seconds_median': statistics.median(totals),
'unescape_seconds_median': statistics.median(unescapes),
'output_sha256': hashlib.sha256(output.encode()).hexdigest(),
}


if __name__ == '__main__':
print(json.dumps({'python': sys.version, 'runs': [measure(n) for n in (500, 1000, 2000, 4000)]}, indent=2))
50 changes: 50 additions & 0 deletions test/test_markdown2.py
Original file line number Diff line number Diff line change
Expand Up @@ -220,6 +220,56 @@ class DirectTestCase(_MarkdownTestCase):
Python-markdown (markdown.py).
"""

def test_many_distinct_code_spans(self):
source = '\n\n'.join('`value_%s`' % i for i in range(1000))
expected = '\n\n'.join('<p><code>value_%s</code></p>' % i for i in range(1000)) + '\n'
self.assertEqual(markdown2.markdown(source), expected)

def test_unescape_nested_tokens(self):
md = markdown2.Markdown()
md.reset()
inner = r'\1\g<0>\\ *'
middle = '<code>%s</code>' % markdown2._hash_text(inner)
outer = '<div>%s</div>' % markdown2._hash_text(middle)
for entries in ([(inner, markdown2._hash_text(inner)),
(middle, markdown2._hash_text(middle))],
[(middle, markdown2._hash_text(middle)),
(inner, markdown2._hash_text(inner))]):
md._code_table = dict(entries)
md.html_blocks = {markdown2._hash_text(outer): outer}
self.assertEqual(md._unescape_special_chars(markdown2._hash_text(outer)),
'<div><code>%s</code></div>' % inner)

def test_unescape_special_chars_inside_html(self):
md = markdown2.Markdown()
md.reset()
html = '<div>%s %s</div>' % (md._escape_table['*'], md._escape_table['\\'])
token = markdown2._hash_text(html)
md.html_blocks[token] = html
self.assertEqual(md._unescape_special_chars(token), '<div>* \\</div>')

def test_unescape_leaves_unknown_tokens_and_plain_text(self):
md = markdown2.Markdown()
md.reset()
text = 'plain \\ text <div>md5-%s</div>' % ('0' * 32)
self.assertEqual(md._unescape_special_chars(text), text)

def test_unescape_repeated_tokens(self):
md = markdown2.Markdown()
md.reset()
value = r'\g<0> * literal'
token = markdown2._hash_text(value)
md._code_table[value] = token
self.assertEqual(md._unescape_special_chars('%s %s' % (token, token)), '%s %s' % (value, value))

def test_unescape_duplicate_token_priority(self):
md = markdown2.Markdown()
md.reset()
token = md._escape_table['*']
md._code_table['code value'] = token
md.html_blocks[token] = '<div>HTML value</div>'
self.assertEqual(md._unescape_special_chars(token), '*')

def test_slow_hr(self):
import time
text = """\
Expand Down