#!/usr/bin/env python3 import markdown import sys import io import re from pygments.formatters import HtmlFormatter from markdown.extensions.toc import TocExtension from markdown.extensions import Extension from markdown.preprocessors import Preprocessor # Python-Markdown (unlike CommonMark/GitHub) refuses to let a list interrupt a # paragraph unless there's a blank line between them, so a very common README # shape renders as one run-on paragraph instead of a list: # # Passing: # - foo # - bar # # This preprocessor restores the GitHub behavior: when a bullet ( - * + ) or a # "1."/"1)" ordered-list marker immediately follows a paragraph line, insert a # blank line so the list starts. We deliberately only interrupt on "1" for # ordered lists (matching CommonMark) so prose like "... the year\n1985. was # ..." is not turned into a list. Fenced code, indented code, and lines that # already follow a list item are left untouched. _MD_FENCE = re.compile(r'^(\s{0,3})(`{3,}|~{3,})') _MD_BULLET = re.compile(r'^ {0,3}[-*+][ \t]+\S') _MD_ORDERED1 = re.compile(r'^ {0,3}1[.)][ \t]+\S') _MD_ANYITEM = re.compile(r'^ {0,3}(?:[-*+]|\d{1,9}[.)])[ \t]') class _ListInterruptPreprocessor(Preprocessor): def run(self, lines): out, in_fence, fence = [], False, None # Track whether we're already inside a list. A bullet that merely # continues an existing list must NOT get a blank line inserted before # it: that would turn a tight list into a loose one (every item wrapped # in
), which is exactly what happens when a preceding item wraps onto
# an indented continuation line and we mistake the next bullet for a
# paragraph->list transition. We only want to interrupt a real paragraph.
in_list = False
for line in lines:
m = _MD_FENCE.match(line)
if m:
tok = m.group(2)[0]
if not in_fence:
in_fence, fence = True, tok
elif fence == tok:
in_fence, fence = False, None
out.append(line)
continue
if in_fence:
out.append(line)
continue
if not line.strip():
# a blank line doesn't end a list (loose lists / lazy
# continuation), so leave in_list untouched
out.append(line)
continue
is_item = bool(_MD_ANYITEM.match(line))
indented = line[:1] in (' ', '\t')
# a non-indented, non-item line after a list closes that list
if in_list and not is_item and not indented:
in_list = False
if out:
prev = out[-1]
if (prev.strip() and not in_list
and not _MD_ANYITEM.match(prev)
and not prev.startswith((' ', '\t'))
and (_MD_BULLET.match(line) or _MD_ORDERED1.match(line))):
out.append('')
if is_item:
in_list = True
out.append(line)
return out
class ListInterruptExtension(Extension):
def extendMarkdown(self, md):
# priority 21: just above the built-in 'normalize_whitespace' (20) so we
# operate on clean lines before block parsing.
md.preprocessors.register(
_ListInterruptPreprocessor(md), 'list_interrupt', 21)
sys.stdin = io.TextIOWrapper(sys.stdin.buffer, encoding='utf-8')
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8')
sys.stdout.write('''
''')
# Render to a string (not straight to stdout) so we can post-process it.
# Note: you may want to run this through bleach for sanitization
html = markdown.markdown(
sys.stdin.read(),
output_format="html5",
extensions=[
"markdown.extensions.fenced_code",
"markdown.extensions.codehilite",
"markdown.extensions.tables",
"markdown.extensions.sane_lists",
ListInterruptExtension(),
TocExtension(anchorlink=True)],
extension_configs={
"markdown.extensions.codehilite":{"css_class":"highlight"}})
# A README's relative paths point at repo content, but a browser resolves them
# against the page the readme is shown on (/ -> /
]*?\bsrc=")([^"]*)(")', rewrite_img, html, flags=re.I)
html = re.sub(r'(]*?\bhref=")([^"]*)(")', rewrite_href, html, flags=re.I)
sys.stdout.write("