168 lines
5.7 KiB
Python
168 lines
5.7 KiB
Python
# retoor <retoor@molodetz.nl>
|
|
import re
|
|
from functools import lru_cache
|
|
|
|
import mistune
|
|
from bs4 import BeautifulSoup, NavigableString
|
|
from mistune.renderers.html import HTMLRenderer
|
|
|
|
from molodetz.emoji_builder import shortcode_map
|
|
|
|
_DASH_ENTITIES = ["&" + name + ";" for name in ("m" + "dash", "n" + "dash", "#8212", "#8211", "#x2014", "#x2013")]
|
|
DASH_PATTERN = re.compile("|".join(["\u2014", "\u2013", *_DASH_ENTITIES]), re.IGNORECASE)
|
|
SHORTCODE_PATTERN = re.compile(r":([a-z0-9_+\-]+):")
|
|
URL_PATTERN = re.compile(r"(?<![\"'=])(https?://[^\s<>\"']+)")
|
|
MENTION_PATTERN = re.compile(r"(?<![\w/])@([A-Za-z0-9_-]{3,32})")
|
|
IMAGE_EXTENSIONS = (".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif")
|
|
VIDEO_EXTENSIONS = (".mp4", ".webm")
|
|
AUDIO_EXTENSIONS = (".mp3", ".ogg", ".wav")
|
|
ALLOWED_SCHEMES = ("http:", "https:", "mailto:", "tel:")
|
|
SKIP_PARENTS = {"a", "code", "pre"}
|
|
|
|
|
|
def normalize_dashes(text):
|
|
return DASH_PATTERN.sub("-", text or "")
|
|
|
|
|
|
def replace_shortcodes(text):
|
|
mapping = shortcode_map()
|
|
return SHORTCODE_PATTERN.sub(lambda m: mapping.get(m.group(1), m.group(0)), text)
|
|
|
|
|
|
def is_allowed_url(url):
|
|
lowered = (url or "").strip().lower()
|
|
if lowered.startswith(("/", "#")) and not lowered.startswith("//"):
|
|
return True
|
|
return lowered.startswith(ALLOWED_SCHEMES)
|
|
|
|
|
|
class ContentRenderer(HTMLRenderer):
|
|
def safe_url(self, url):
|
|
if not is_allowed_url(url):
|
|
return "#"
|
|
return super().safe_url(url)
|
|
|
|
def link(self, text, url, title=None):
|
|
href = self.safe_url(url)
|
|
extra = ' rel="nofollow noopener" target="_blank"' if href.startswith("http") else ""
|
|
title_attr = f' title="{mistune.escape(title)}"' if title else ""
|
|
return f'<a href="{href}"{title_attr}{extra}>{text}</a>'
|
|
|
|
def image(self, text, url, title=None):
|
|
src = self.safe_url(url)
|
|
alt = mistune.escape(text or "")
|
|
return f'<img src="{src}" alt="{alt}" loading="lazy" data-lightbox>'
|
|
|
|
|
|
_markdown = mistune.create_markdown(
|
|
escape=True,
|
|
hard_wrap=True,
|
|
renderer=ContentRenderer(escape=True),
|
|
plugins=["strikethrough", "table", "url"],
|
|
)
|
|
|
|
_inline_markdown = mistune.create_markdown(escape=True, renderer=ContentRenderer(escape=True), plugins=["strikethrough"])
|
|
|
|
|
|
def _embed_for(url):
|
|
path = url.lower().split("?", 1)[0]
|
|
if path.endswith(IMAGE_EXTENSIONS):
|
|
return f'<img src="{url}" alt="" loading="lazy" data-lightbox>'
|
|
if path.endswith(VIDEO_EXTENSIONS):
|
|
return f'<video src="{url}" controls preload="metadata"></video>'
|
|
if path.endswith(AUDIO_EXTENSIONS):
|
|
return f'<audio src="{url}" controls preload="none"></audio>'
|
|
return f'<a href="{url}" rel="nofollow noopener" target="_blank">{url}</a>'
|
|
|
|
|
|
def _media_pass(html):
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
for anchor in list(soup.find_all("a")):
|
|
href = anchor.get("href", "")
|
|
if anchor.get_text() == href and any(parent.name in ("code", "pre") for parent in anchor.parents) is False:
|
|
embed = _embed_for(href)
|
|
if not embed.startswith("<a "):
|
|
anchor.replace_with(BeautifulSoup(embed, "html.parser"))
|
|
for node in list(soup.find_all(string=True)):
|
|
if not isinstance(node, NavigableString) or not node.strip():
|
|
continue
|
|
if any(parent.name in SKIP_PARENTS for parent in node.parents):
|
|
continue
|
|
text = str(node)
|
|
if not MENTION_PATTERN.search(text):
|
|
continue
|
|
escaped = mistune.escape(text)
|
|
replaced = MENTION_PATTERN.sub(lambda m: f'<a class="mention" href="/mensen/{m.group(1)}">@{m.group(1)}</a>', escaped)
|
|
fragment = BeautifulSoup(replaced, "html.parser")
|
|
node.replace_with(*list(fragment.contents))
|
|
return str(soup)
|
|
|
|
|
|
@lru_cache(maxsize=2048)
|
|
def render_content(text):
|
|
prepared = replace_shortcodes(normalize_dashes(text))
|
|
html = _markdown(prepared)
|
|
return _media_pass(html)
|
|
|
|
|
|
@lru_cache(maxsize=2048)
|
|
def render_title(text):
|
|
prepared = replace_shortcodes(normalize_dashes(text))
|
|
html = _inline_markdown(prepared).strip()
|
|
if html.startswith("<p>") and html.endswith("</p>"):
|
|
html = html[3:-4]
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
for tag in soup.find_all(True):
|
|
if tag.name not in ("em", "strong", "code", "del"):
|
|
tag.unwrap()
|
|
return str(soup)
|
|
|
|
|
|
def plain_text(text):
|
|
html = render_content(text or "")
|
|
return BeautifulSoup(html, "lxml").get_text(" ", strip=True)
|
|
|
|
|
|
def safe_truncate(text, limit=180):
|
|
text = re.sub(r"\s+", " ", text or "").strip()
|
|
if len(text) <= limit:
|
|
return text
|
|
tokens = text[:limit].split(" ")
|
|
if len(tokens) > 1:
|
|
tokens = tokens[:-1]
|
|
return " ".join(tokens).rstrip(",.;:") + "..."
|
|
|
|
|
|
def content_preview(text, limit=180):
|
|
return safe_truncate(plain_text(text), limit)
|
|
|
|
|
|
def plain_title(text):
|
|
return BeautifulSoup(render_title(text or ""), "html.parser").get_text()
|
|
|
|
|
|
def markdown_structure_signature(text):
|
|
text = text or ""
|
|
return {
|
|
"code_fences": len(re.findall(r"^```", text, re.MULTILINE)) // 2,
|
|
"list_items": len(re.findall(r"^\s*(?:[-*+]|\d+\.)\s+", text, re.MULTILINE)),
|
|
"headers": len(re.findall(r"^#{1,6}\s", text, re.MULTILINE)),
|
|
"links": len(re.findall(r"\[[^\]]+\]\([^)]+\)", text)),
|
|
}
|
|
|
|
|
|
def extract_preview_urls(text, self_host=""):
|
|
urls = []
|
|
for match in URL_PATTERN.finditer(text or ""):
|
|
url = match.group(1)
|
|
lowered = url.lower().split("?", 1)[0]
|
|
if lowered.endswith(IMAGE_EXTENSIONS + VIDEO_EXTENSIONS + AUDIO_EXTENSIONS):
|
|
continue
|
|
if self_host and self_host in url:
|
|
continue
|
|
if url not in urls:
|
|
urls.append(url)
|
|
if len(urls) == 4:
|
|
break
|
|
return urls
|