Files
ad/molodetz/rendering.py
T

168 lines
5.7 KiB
Python

# retoor <retoor@molodetz.nl>
import re
from functools import lru_cache
import mistune
from bs4 import BeautifulSoup, NavigableString
from mistune.renderers.html import HTMLRenderer
from molodetz.emoji_builder import shortcode_map
_DASH_ENTITIES = ["&" + name + ";" for name in ("m" + "dash", "n" + "dash", "#8212", "#8211", "#x2014", "#x2013")]
DASH_PATTERN = re.compile("|".join(["\u2014", "\u2013", *_DASH_ENTITIES]), re.IGNORECASE)
SHORTCODE_PATTERN = re.compile(r":([a-z0-9_+\-]+):")
URL_PATTERN = re.compile(r"(?<![\"'=])(https?://[^\s<>\"']+)")
MENTION_PATTERN = re.compile(r"(?<![\w/])@([A-Za-z0-9_-]{3,32})")
IMAGE_EXTENSIONS = (".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif")
VIDEO_EXTENSIONS = (".mp4", ".webm")
AUDIO_EXTENSIONS = (".mp3", ".ogg", ".wav")
ALLOWED_SCHEMES = ("http:", "https:", "mailto:", "tel:")
SKIP_PARENTS = {"a", "code", "pre"}
def normalize_dashes(text):
return DASH_PATTERN.sub("-", text or "")
def replace_shortcodes(text):
mapping = shortcode_map()
return SHORTCODE_PATTERN.sub(lambda m: mapping.get(m.group(1), m.group(0)), text)
def is_allowed_url(url):
lowered = (url or "").strip().lower()
if lowered.startswith(("/", "#")) and not lowered.startswith("//"):
return True
return lowered.startswith(ALLOWED_SCHEMES)
class ContentRenderer(HTMLRenderer):
def safe_url(self, url):
if not is_allowed_url(url):
return "#"
return super().safe_url(url)
def link(self, text, url, title=None):
href = self.safe_url(url)
extra = ' rel="nofollow noopener" target="_blank"' if href.startswith("http") else ""
title_attr = f' title="{mistune.escape(title)}"' if title else ""
return f'<a href="{href}"{title_attr}{extra}>{text}</a>'
def image(self, text, url, title=None):
src = self.safe_url(url)
alt = mistune.escape(text or "")
return f'<img src="{src}" alt="{alt}" loading="lazy" data-lightbox>'
_markdown = mistune.create_markdown(
escape=True,
hard_wrap=True,
renderer=ContentRenderer(escape=True),
plugins=["strikethrough", "table", "url"],
)
_inline_markdown = mistune.create_markdown(escape=True, renderer=ContentRenderer(escape=True), plugins=["strikethrough"])
def _embed_for(url):
path = url.lower().split("?", 1)[0]
if path.endswith(IMAGE_EXTENSIONS):
return f'<img src="{url}" alt="" loading="lazy" data-lightbox>'
if path.endswith(VIDEO_EXTENSIONS):
return f'<video src="{url}" controls preload="metadata"></video>'
if path.endswith(AUDIO_EXTENSIONS):
return f'<audio src="{url}" controls preload="none"></audio>'
return f'<a href="{url}" rel="nofollow noopener" target="_blank">{url}</a>'
def _media_pass(html):
soup = BeautifulSoup(html, "html.parser")
for anchor in list(soup.find_all("a")):
href = anchor.get("href", "")
if anchor.get_text() == href and any(parent.name in ("code", "pre") for parent in anchor.parents) is False:
embed = _embed_for(href)
if not embed.startswith("<a "):
anchor.replace_with(BeautifulSoup(embed, "html.parser"))
for node in list(soup.find_all(string=True)):
if not isinstance(node, NavigableString) or not node.strip():
continue
if any(parent.name in SKIP_PARENTS for parent in node.parents):
continue
text = str(node)
if not MENTION_PATTERN.search(text):
continue
escaped = mistune.escape(text)
replaced = MENTION_PATTERN.sub(lambda m: f'<a class="mention" href="/mensen/{m.group(1)}">@{m.group(1)}</a>', escaped)
fragment = BeautifulSoup(replaced, "html.parser")
node.replace_with(*list(fragment.contents))
return str(soup)
@lru_cache(maxsize=2048)
def render_content(text):
prepared = replace_shortcodes(normalize_dashes(text))
html = _markdown(prepared)
return _media_pass(html)
@lru_cache(maxsize=2048)
def render_title(text):
prepared = replace_shortcodes(normalize_dashes(text))
html = _inline_markdown(prepared).strip()
if html.startswith("<p>") and html.endswith("</p>"):
html = html[3:-4]
soup = BeautifulSoup(html, "html.parser")
for tag in soup.find_all(True):
if tag.name not in ("em", "strong", "code", "del"):
tag.unwrap()
return str(soup)
def plain_text(text):
html = render_content(text or "")
return BeautifulSoup(html, "lxml").get_text(" ", strip=True)
def safe_truncate(text, limit=180):
text = re.sub(r"\s+", " ", text or "").strip()
if len(text) <= limit:
return text
tokens = text[:limit].split(" ")
if len(tokens) > 1:
tokens = tokens[:-1]
return " ".join(tokens).rstrip(",.;:") + "..."
def content_preview(text, limit=180):
return safe_truncate(plain_text(text), limit)
def plain_title(text):
return BeautifulSoup(render_title(text or ""), "html.parser").get_text()
def markdown_structure_signature(text):
text = text or ""
return {
"code_fences": len(re.findall(r"^```", text, re.MULTILINE)) // 2,
"list_items": len(re.findall(r"^\s*(?:[-*+]|\d+\.)\s+", text, re.MULTILINE)),
"headers": len(re.findall(r"^#{1,6}\s", text, re.MULTILINE)),
"links": len(re.findall(r"\[[^\]]+\]\([^)]+\)", text)),
}
def extract_preview_urls(text, self_host=""):
urls = []
for match in URL_PATTERN.finditer(text or ""):
url = match.group(1)
lowered = url.lower().split("?", 1)[0]
if lowered.endswith(IMAGE_EXTENSIONS + VIDEO_EXTENSIONS + AUDIO_EXTENSIONS):
continue
if self_host and self_host in url:
continue
if url not in urls:
urls.append(url)
if len(urls) == 4:
break
return urls