Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 39 additions & 1 deletion apps/core/markdown.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,12 @@
import xml.etree.ElementTree as etree
from html import escape
from html.parser import HTMLParser
from urllib.parse import urlparse

import markdown as md_lib
from markdown.extensions import Extension
from markdown.inlinepatterns import InlineProcessor
from markdown.util import AtomicString


ALLOWED_TAGS = {
Expand Down Expand Up @@ -65,8 +69,42 @@ def _append_starttag(self, tag, attrs):
self.parts.append(f'<{tag}{"".join(clean_attrs)}>')


_BARE_URL_RE = r'https?://[^\s<>]+'
_BARE_URL_TRAILING_PUNCTUATION = '.,;:!?\'"'


class _BareUrlInlineProcessor(InlineProcessor):
"""Turn bare URLs (not already part of `[text](url)` or `<url>` markup) into links."""

def handleMatch(self, m, data):
url = m.group(0)
end = m.end(0)
# Trailing punctuation is usually sentence punctuation, not part of the URL.
# An unbalanced closing paren is treated the same way, so a URL in
# "(see https://example.com)" doesn't swallow the closing paren.
while url and (
url[-1] in _BARE_URL_TRAILING_PUNCTUATION
or (url[-1] == ')' and url.count('(') < url.count(')'))
):
Comment on lines +85 to +88
url = url[:-1]
end -= 1
if not url:
return None, None, None
el = etree.Element('a')
el.set('href', url)
el.text = AtomicString(url)
return el, m.start(0), end


class _AutolinkExtension(Extension):
def extendMarkdown(self, md):
# Priority 75 runs after 'link'/'autolink'/'html' (which claim URLs already
# wrapped in markdown or HTML link syntax) but before emphasis patterns.
md.inlinePatterns.register(_BareUrlInlineProcessor(_BARE_URL_RE, md), 'autolink_bare_url', 75)


def render_markdown(text: str) -> str:
html = md_lib.markdown(text or '', extensions=['fenced_code', 'tables'])
html = md_lib.markdown(text or '', extensions=['fenced_code', 'tables', _AutolinkExtension()])
sanitizer = _MarkdownSanitizer()
sanitizer.feed(html)
sanitizer.close()
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{% comment %}
Markdown quick-reference shown beside the entry form.
Keep rows in sync with apps/core/markdown.py (render_markdown):
fenced_code + tables, headings h1-h3, no images.
fenced_code + tables, headings h1-h3, no images, bare URLs auto-link.
{% endcomment %}
<div class="bg-white rounded-lg border border-slate-200 shadow-sm overflow-hidden">

Expand All @@ -27,6 +27,10 @@ <h2 class="text-sm font-semibold text-slate-700">Markdown cheat sheet</h2>
<dt class="text-xs text-slate-500 shrink-0">Link</dt>
<dd class="font-mono text-xs text-slate-700 bg-slate-50 rounded px-1.5 py-0.5">[text](https://…)</dd>
</div>
<div class="px-4 py-2 flex items-center justify-between gap-3">
<dt class="text-xs text-slate-500 shrink-0">Auto-link</dt>
<dd class="font-mono text-xs text-slate-700 bg-slate-50 rounded px-1.5 py-0.5">https://…</dd>
</div>
<div class="px-4 py-2 flex items-center justify-between gap-3">
<dt class="text-xs text-slate-500 shrink-0">Bullet list</dt>
<dd class="font-mono text-xs text-slate-700 bg-slate-50 rounded px-1.5 py-0.5">- item</dd>
Expand Down
29 changes: 29 additions & 0 deletions tests/test_entries.py
Original file line number Diff line number Diff line change
Expand Up @@ -427,6 +427,35 @@ def test_empty_description_returns_empty(self, client, user):
resp = client.post(reverse('entries:markdown-preview'), {'description': ''})
assert resp.status_code == 200

def test_autolinks_bare_url(self, client, user):
client.force_login(user)
resp = client.post(reverse('entries:markdown-preview'), {
'description': 'See https://cds.cern.ch/record/2967685 for details.',
})
assert resp.status_code == 200
assert (
b'<a href="https://cds.cern.ch/record/2967685">'
b'https://cds.cern.ch/record/2967685</a>' in resp.content
)

def test_autolink_does_not_swallow_trailing_punctuation(self, client, user):
client.force_login(user)
resp = client.post(reverse('entries:markdown-preview'), {
'description': 'Visit https://example.com, then https://example.com/other.',
})
assert resp.status_code == 200
assert b'<a href="https://example.com">https://example.com</a>,' in resp.content
assert b'<a href="https://example.com/other">https://example.com/other</a>.' in resp.content

Comment on lines +441 to +449
def test_autolink_does_not_duplicate_existing_markdown_link(self, client, user):
client.force_login(user)
resp = client.post(reverse('entries:markdown-preview'), {
'description': '[the paper](https://example.com/already)',
})
assert resp.status_code == 200
assert resp.content.count(b'<a ') == 1
assert b'<a href="https://example.com/already">the paper</a>' in resp.content


# ── Entry description templates ───────────────────────────────────────────────

Expand Down