Molodetz blogrol volgens DPP-template
This commit is contained in:
@@ -0,0 +1,167 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
import re
|
||||
from functools import lru_cache
|
||||
|
||||
import mistune
|
||||
from bs4 import BeautifulSoup, NavigableString
|
||||
from mistune.renderers.html import HTMLRenderer
|
||||
|
||||
from molodetz.emoji_builder import shortcode_map
|
||||
|
||||
_DASH_ENTITIES = ["&" + name + ";" for name in ("m" + "dash", "n" + "dash", "#8212", "#8211", "#x2014", "#x2013")]
|
||||
DASH_PATTERN = re.compile("|".join(["\u2014", "\u2013", *_DASH_ENTITIES]), re.IGNORECASE)
|
||||
SHORTCODE_PATTERN = re.compile(r":([a-z0-9_+\-]+):")
|
||||
URL_PATTERN = re.compile(r"(?<![\"'=])(https?://[^\s<>\"']+)")
|
||||
MENTION_PATTERN = re.compile(r"(?<![\w/])@([A-Za-z0-9_-]{3,32})")
|
||||
IMAGE_EXTENSIONS = (".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif")
|
||||
VIDEO_EXTENSIONS = (".mp4", ".webm")
|
||||
AUDIO_EXTENSIONS = (".mp3", ".ogg", ".wav")
|
||||
ALLOWED_SCHEMES = ("http:", "https:", "mailto:", "tel:")
|
||||
SKIP_PARENTS = {"a", "code", "pre"}
|
||||
|
||||
|
||||
def normalize_dashes(text):
|
||||
return DASH_PATTERN.sub("-", text or "")
|
||||
|
||||
|
||||
def replace_shortcodes(text):
|
||||
mapping = shortcode_map()
|
||||
return SHORTCODE_PATTERN.sub(lambda m: mapping.get(m.group(1), m.group(0)), text)
|
||||
|
||||
|
||||
def is_allowed_url(url):
|
||||
lowered = (url or "").strip().lower()
|
||||
if lowered.startswith(("/", "#")) and not lowered.startswith("//"):
|
||||
return True
|
||||
return lowered.startswith(ALLOWED_SCHEMES)
|
||||
|
||||
|
||||
class ContentRenderer(HTMLRenderer):
|
||||
def safe_url(self, url):
|
||||
if not is_allowed_url(url):
|
||||
return "#"
|
||||
return super().safe_url(url)
|
||||
|
||||
def link(self, text, url, title=None):
|
||||
href = self.safe_url(url)
|
||||
extra = ' rel="nofollow noopener" target="_blank"' if href.startswith("http") else ""
|
||||
title_attr = f' title="{mistune.escape(title)}"' if title else ""
|
||||
return f'<a href="{href}"{title_attr}{extra}>{text}</a>'
|
||||
|
||||
def image(self, text, url, title=None):
|
||||
src = self.safe_url(url)
|
||||
alt = mistune.escape(text or "")
|
||||
return f'<img src="{src}" alt="{alt}" loading="lazy" data-lightbox>'
|
||||
|
||||
|
||||
_markdown = mistune.create_markdown(
|
||||
escape=True,
|
||||
hard_wrap=True,
|
||||
renderer=ContentRenderer(escape=True),
|
||||
plugins=["strikethrough", "table", "url"],
|
||||
)
|
||||
|
||||
_inline_markdown = mistune.create_markdown(escape=True, renderer=ContentRenderer(escape=True), plugins=["strikethrough"])
|
||||
|
||||
|
||||
def _embed_for(url):
|
||||
path = url.lower().split("?", 1)[0]
|
||||
if path.endswith(IMAGE_EXTENSIONS):
|
||||
return f'<img src="{url}" alt="" loading="lazy" data-lightbox>'
|
||||
if path.endswith(VIDEO_EXTENSIONS):
|
||||
return f'<video src="{url}" controls preload="metadata"></video>'
|
||||
if path.endswith(AUDIO_EXTENSIONS):
|
||||
return f'<audio src="{url}" controls preload="none"></audio>'
|
||||
return f'<a href="{url}" rel="nofollow noopener" target="_blank">{url}</a>'
|
||||
|
||||
|
||||
def _media_pass(html):
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
for anchor in list(soup.find_all("a")):
|
||||
href = anchor.get("href", "")
|
||||
if anchor.get_text() == href and any(parent.name in ("code", "pre") for parent in anchor.parents) is False:
|
||||
embed = _embed_for(href)
|
||||
if not embed.startswith("<a "):
|
||||
anchor.replace_with(BeautifulSoup(embed, "html.parser"))
|
||||
for node in list(soup.find_all(string=True)):
|
||||
if not isinstance(node, NavigableString) or not node.strip():
|
||||
continue
|
||||
if any(parent.name in SKIP_PARENTS for parent in node.parents):
|
||||
continue
|
||||
text = str(node)
|
||||
if not MENTION_PATTERN.search(text):
|
||||
continue
|
||||
escaped = mistune.escape(text)
|
||||
replaced = MENTION_PATTERN.sub(lambda m: f'<a class="mention" href="/mensen/{m.group(1)}">@{m.group(1)}</a>', escaped)
|
||||
fragment = BeautifulSoup(replaced, "html.parser")
|
||||
node.replace_with(*list(fragment.contents))
|
||||
return str(soup)
|
||||
|
||||
|
||||
@lru_cache(maxsize=2048)
|
||||
def render_content(text):
|
||||
prepared = replace_shortcodes(normalize_dashes(text))
|
||||
html = _markdown(prepared)
|
||||
return _media_pass(html)
|
||||
|
||||
|
||||
@lru_cache(maxsize=2048)
|
||||
def render_title(text):
|
||||
prepared = replace_shortcodes(normalize_dashes(text))
|
||||
html = _inline_markdown(prepared).strip()
|
||||
if html.startswith("<p>") and html.endswith("</p>"):
|
||||
html = html[3:-4]
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
for tag in soup.find_all(True):
|
||||
if tag.name not in ("em", "strong", "code", "del"):
|
||||
tag.unwrap()
|
||||
return str(soup)
|
||||
|
||||
|
||||
def plain_text(text):
|
||||
html = render_content(text or "")
|
||||
return BeautifulSoup(html, "lxml").get_text(" ", strip=True)
|
||||
|
||||
|
||||
def safe_truncate(text, limit=180):
|
||||
text = re.sub(r"\s+", " ", text or "").strip()
|
||||
if len(text) <= limit:
|
||||
return text
|
||||
tokens = text[:limit].split(" ")
|
||||
if len(tokens) > 1:
|
||||
tokens = tokens[:-1]
|
||||
return " ".join(tokens).rstrip(",.;:") + "..."
|
||||
|
||||
|
||||
def content_preview(text, limit=180):
|
||||
return safe_truncate(plain_text(text), limit)
|
||||
|
||||
|
||||
def plain_title(text):
|
||||
return BeautifulSoup(render_title(text or ""), "html.parser").get_text()
|
||||
|
||||
|
||||
def markdown_structure_signature(text):
|
||||
text = text or ""
|
||||
return {
|
||||
"code_fences": len(re.findall(r"^```", text, re.MULTILINE)) // 2,
|
||||
"list_items": len(re.findall(r"^\s*(?:[-*+]|\d+\.)\s+", text, re.MULTILINE)),
|
||||
"headers": len(re.findall(r"^#{1,6}\s", text, re.MULTILINE)),
|
||||
"links": len(re.findall(r"\[[^\]]+\]\([^)]+\)", text)),
|
||||
}
|
||||
|
||||
|
||||
def extract_preview_urls(text, self_host=""):
|
||||
urls = []
|
||||
for match in URL_PATTERN.finditer(text or ""):
|
||||
url = match.group(1)
|
||||
lowered = url.lower().split("?", 1)[0]
|
||||
if lowered.endswith(IMAGE_EXTENSIONS + VIDEO_EXTENSIONS + AUDIO_EXTENSIONS):
|
||||
continue
|
||||
if self_host and self_host in url:
|
||||
continue
|
||||
if url not in urls:
|
||||
urls.append(url)
|
||||
if len(urls) == 4:
|
||||
break
|
||||
return urls
|
||||
Reference in New Issue
Block a user