diff options
Diffstat (limited to 'filters/html-converters/md2html')
| -rwxr-xr-x | filters/html-converters/md2html | 71 |
1 files changed, 60 insertions, 11 deletions
diff --git a/filters/html-converters/md2html b/filters/html-converters/md2html index faa71f3..90021a7 100755 --- a/filters/html-converters/md2html +++ b/filters/html-converters/md2html @@ -357,27 +357,76 @@ html = markdown.markdown( extension_configs={ "markdown.extensions.codehilite":{"css_class":"highlight"}}) -# A README's relative image paths (e.g. screenshots/x.png) are repo blobs, which -# cgit only serves via its `plain` command; but a browser resolves a relative -# <img src> against the about-page URL (/<repo>/about/), so the image 404s. -# Rewrite relative <img src> to an absolute /<repo>/plain/<path>?h=<branch> URL -# (absolute so it also works where cgit renders the readme on the summary page). +# A README's relative paths point at repo content, but a browser resolves them +# against the page the readme is shown on (/<repo>/about/ or the summary page +# /<repo>/), not against the repo root — so they 404. cgit serves blobs only via +# its `plain`/`tree` commands, so rewrite relative references to absolute cgit +# URLs (absolute so it works both on the about page and the summary page): +# <img src> -> /<repo>/plain/<path>?h=<branch> (raw bytes for the browser) +# <a href> -> /<repo>/tree/<path>?h=<branch> (browsable file/dir view) +# Relative hrefs are resolved as paths against the repo root; a href that climbs +# out of the repo with ".." (e.g. ../sibling) is resolved against the repo's URL +# location instead (-> /sibling), matching how sibling repos are addressed. # cgit exports the repo context to about-filters via these environment vars; if -# they're absent we leave the src untouched. +# they're absent we leave the references untouched. import os, re +from urllib.parse import urljoin repo = os.environ.get("CGIT_REPO_URL", "").strip("/") if repo: branch = os.environ.get("CGIT_REPO_DEFBRANCH", "").strip() - plain = "/" + repo + "/plain/" + repo_root = "/" + repo + "/" + plain = repo_root + "plain/" query = ("?h=" + branch) if branch else "" - # leave absolute, protocol-relative, root-relative, anchor and data: URLs alone - skip = re.compile(r"^(?:[a-z][a-z0-9+.\-]*:|//|/|#)", re.I) - def rewrite(m): + # leave absolute, protocol-relative, root-relative, anchor, query-only and + # scheme (mailto:, http:, data:, ...) URLs alone + skip = re.compile(r"^(?:[a-z][a-z0-9+.\-]*:|//|/|[#?])", re.I) + + def rewrite_img(m): url = m.group(2) if not url or skip.match(url): return m.group(0) return m.group(1) + plain + re.sub(r"^\./", "", url) + query + m.group(3) - html = re.sub(r'(<img\b[^>]*?\bsrc=")([^"]*)(")', rewrite, html, flags=re.I) + + def _split_frag(url): + # split off a trailing #fragment (kept verbatim on the rewritten URL) + if "#" in url: + path, frag = url.split("#", 1) + return path, "#" + frag + return url, "" + + def _normalize(path): + # resolve "." / ".." against the repo root; returns (climbs, cleaned) + # where `climbs` counts ".." segments that escaped above the repo root. + out, climbs = [], 0 + for seg in path.split("/"): + if seg in ("", "."): + continue + if seg == "..": + if out: + out.pop() + else: + climbs += 1 + else: + out.append(seg) + return climbs, "/".join(out) + + def rewrite_href(m): + url = m.group(2) + if not url or skip.match(url): + return m.group(0) + path, frag = _split_frag(url) + climbs, cleaned = _normalize(path) + if climbs: + # escapes the repo (sibling repo etc.): resolve against repo location + dest = urljoin(repo_root, "../" * climbs + cleaned) + elif cleaned: + dest = repo_root + "tree/" + cleaned + query + else: + dest = repo_root + (query or "") + return m.group(1) + dest + frag + m.group(3) + + html = re.sub(r'(<img\b[^>]*?\bsrc=")([^"]*)(")', rewrite_img, html, flags=re.I) + html = re.sub(r'(<a\b[^>]*?\bhref=")([^"]*)(")', rewrite_href, html, flags=re.I) sys.stdout.write("<div class='markdown-body'>") sys.stdout.write(html) |
