# retoor import re from functools import lru_cache import mistune from bs4 import BeautifulSoup, NavigableString from mistune.renderers.html import HTMLRenderer from molodetz.emoji_builder import shortcode_map _DASH_ENTITIES = ["&" + name + ";" for name in ("m" + "dash", "n" + "dash", "#8212", "#8211", "#x2014", "#x2013")] DASH_PATTERN = re.compile("|".join(["\u2014", "\u2013", *_DASH_ENTITIES]), re.IGNORECASE) SHORTCODE_PATTERN = re.compile(r":([a-z0-9_+\-]+):") URL_PATTERN = re.compile(r"(?\"']+)") MENTION_PATTERN = re.compile(r"(?{text}' def image(self, text, url, title=None): src = self.safe_url(url) alt = mistune.escape(text or "") return f'{alt}' _markdown = mistune.create_markdown( escape=True, hard_wrap=True, renderer=ContentRenderer(escape=True), plugins=["strikethrough", "table", "url"], ) _inline_markdown = mistune.create_markdown(escape=True, renderer=ContentRenderer(escape=True), plugins=["strikethrough"]) def _embed_for(url): path = url.lower().split("?", 1)[0] if path.endswith(IMAGE_EXTENSIONS): return f'' if path.endswith(VIDEO_EXTENSIONS): return f'' if path.endswith(AUDIO_EXTENSIONS): return f'' return f'{url}' def _media_pass(html): soup = BeautifulSoup(html, "html.parser") for anchor in list(soup.find_all("a")): href = anchor.get("href", "") if anchor.get_text() == href and any(parent.name in ("code", "pre") for parent in anchor.parents) is False: embed = _embed_for(href) if not embed.startswith("@{m.group(1)}', escaped) fragment = BeautifulSoup(replaced, "html.parser") node.replace_with(*list(fragment.contents)) return str(soup) @lru_cache(maxsize=2048) def render_content(text): prepared = replace_shortcodes(normalize_dashes(text)) html = _markdown(prepared) return _media_pass(html) @lru_cache(maxsize=2048) def render_title(text): prepared = replace_shortcodes(normalize_dashes(text)) html = _inline_markdown(prepared).strip() if html.startswith("

") and html.endswith("

"): html = html[3:-4] soup = BeautifulSoup(html, "html.parser") for tag in soup.find_all(True): if tag.name not in ("em", "strong", "code", "del"): tag.unwrap() return str(soup) def plain_text(text): html = render_content(text or "") return BeautifulSoup(html, "lxml").get_text(" ", strip=True) def safe_truncate(text, limit=180): text = re.sub(r"\s+", " ", text or "").strip() if len(text) <= limit: return text tokens = text[:limit].split(" ") if len(tokens) > 1: tokens = tokens[:-1] return " ".join(tokens).rstrip(",.;:") + "..." def content_preview(text, limit=180): return safe_truncate(plain_text(text), limit) def plain_title(text): return BeautifulSoup(render_title(text or ""), "html.parser").get_text() def markdown_structure_signature(text): text = text or "" return { "code_fences": len(re.findall(r"^```", text, re.MULTILINE)) // 2, "list_items": len(re.findall(r"^\s*(?:[-*+]|\d+\.)\s+", text, re.MULTILINE)), "headers": len(re.findall(r"^#{1,6}\s", text, re.MULTILINE)), "links": len(re.findall(r"\[[^\]]+\]\([^)]+\)", text)), } def extract_preview_urls(text, self_host=""): urls = [] for match in URL_PATTERN.finditer(text or ""): url = match.group(1) lowered = url.lower().split("?", 1)[0] if lowered.endswith(IMAGE_EXTENSIONS + VIDEO_EXTENSIONS + AUDIO_EXTENSIONS): continue if self_host and self_host in url: continue if url not in urls: urls.append(url) if len(urls) == 4: break return urls