# retoor import logging import defusedxml.ElementTree as SafeET import httpx from bs4 import BeautifulSoup from molodetz.database import now_iso, update_join_request from molodetz.net_guard import UnsafeURL, guard_public_url from molodetz.stealth import detect_consent_gate, guarded_async_client logger = logging.getLogger(__name__) MAX_BYTES = 512 * 1024 def extract_title(content_type, body): if "xml" in content_type and "html" not in content_type: root = SafeET.fromstring(body) for element in root.iter(): if element.tag.split("}")[-1] == "title" and element.text: return element.text.strip()[:200] return None soup = BeautifulSoup(body, "lxml") if soup.title and soup.title.string: return soup.title.string.strip()[:200] heading = soup.find("h1") return heading.get_text(strip=True)[:200] if heading else None async def check_repo_link(row): url = row.get("repo_url") if not url: return {"status": "none"} try: await guard_public_url(url) async with guarded_async_client(timeout=10.0) as client: response = await client.get(url) body = response.content[:MAX_BYTES] redirect = detect_consent_gate(body.decode("utf-8", "ignore")) if redirect: response = await client.get(redirect) body = response.content[:MAX_BYTES] title = extract_title(response.headers.get("content-type", ""), body) if response.status_code < 400 else None status = f"http {response.status_code}" except UnsafeURL as exc: title, status = None, f"refused: {exc}" except (httpx.HTTPError, OSError, SafeET.ParseError) as exc: title, status = None, f"error: {type(exc).__name__}" except Exception as exc: logger.warning("repo link check failed: %s", exc) title, status = None, f"error: {type(exc).__name__}" update_join_request(row["uid"], link_title=title, link_status=status, link_checked_at=now_iso()) return {"status": status, "title": title}