2026-10-05 09:36:20 +02:00
|
|
|
# retoor <retoor@molodetz.nl>
|
|
|
|
|
import logging
|
|
|
|
|
from xml.sax.saxutils import escape
|
|
|
|
|
|
|
|
|
|
import defusedxml.ElementTree as SafeET
|
|
|
|
|
from fastapi import APIRouter, Request
|
|
|
|
|
from fastapi.responses import PlainTextResponse, Response
|
|
|
|
|
|
|
|
|
|
from molodetz import config
|
|
|
|
|
from molodetz.cache import TTLCache
|
|
|
|
|
from molodetz.constants import TOPICS
|
|
|
|
|
from molodetz.database import db, public_people
|
|
|
|
|
from molodetz.docs_prose import DOCS_PAGES
|
|
|
|
|
from molodetz.seo import base_url
|
|
|
|
|
|
|
|
|
|
router = APIRouter()
|
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
_sitemap_cache = TTLCache(ttl=config.SITEMAP_TTL, max_size=4)
|
|
|
|
|
PER_TABLE_CAP = 5000
|
2026-10-05 11:25:41 +02:00
|
|
|
STATIC_PATHS = ("/", "/flyers", "/memes", "/people", "/join", "/terms", "/privacy", "/docs")
|
2026-10-05 09:36:20 +02:00
|
|
|
|
|
|
|
|
|
|
|
|
|
@router.get("/robots.txt")
|
|
|
|
|
async def robots(request: Request):
|
|
|
|
|
base = base_url(request)
|
|
|
|
|
body = "\n".join(
|
|
|
|
|
[
|
|
|
|
|
"User-agent: *",
|
|
|
|
|
"Disallow: /admin",
|
|
|
|
|
"Disallow: /auth",
|
|
|
|
|
"Disallow: /notifications",
|
|
|
|
|
"Disallow: /profile",
|
|
|
|
|
"Allow: /",
|
|
|
|
|
f"Sitemap: {base}/sitemap.xml",
|
|
|
|
|
"",
|
|
|
|
|
]
|
|
|
|
|
)
|
|
|
|
|
return PlainTextResponse(body)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def build_sitemap(base):
|
|
|
|
|
urls = [(path, None) for path in STATIC_PATHS]
|
|
|
|
|
urls += [(f"/{topic}", None) for topic in TOPICS]
|
|
|
|
|
if "posts" in db.tables:
|
|
|
|
|
rows = list(db["posts"].find(deleted_at=None, status="published", order_by=["-published_at"], _limit=PER_TABLE_CAP + 1))
|
|
|
|
|
if len(rows) > PER_TABLE_CAP:
|
|
|
|
|
logger.warning("sitemap truncated posts at %d", PER_TABLE_CAP)
|
|
|
|
|
rows = rows[:PER_TABLE_CAP]
|
|
|
|
|
urls += [(f"/posts/{row['slug']}", row.get("updated_at")) for row in rows]
|
2026-10-05 11:25:41 +02:00
|
|
|
urls += [(f"/people/{person['username']}", None) for person in public_people()[:PER_TABLE_CAP]]
|
2026-10-05 09:36:20 +02:00
|
|
|
urls += [(f"/docs/{page['slug']}", None) for page in DOCS_PAGES if not page.get("admin")]
|
|
|
|
|
parts = ['<?xml version="1.0" encoding="UTF-8"?>', '<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">']
|
|
|
|
|
for path, modified in urls:
|
|
|
|
|
entry = f"<url><loc>{escape(base + path)}</loc>"
|
|
|
|
|
if modified:
|
|
|
|
|
entry += f"<lastmod>{escape(modified[:10])}</lastmod>"
|
|
|
|
|
parts.append(entry + "</url>")
|
|
|
|
|
parts.append("</urlset>")
|
|
|
|
|
xml = "\n".join(parts)
|
|
|
|
|
SafeET.fromstring(xml.encode("utf-8"))
|
|
|
|
|
return xml
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@router.get("/sitemap.xml")
|
|
|
|
|
async def sitemap(request: Request):
|
|
|
|
|
base = base_url(request)
|
|
|
|
|
xml = _sitemap_cache.get(base)
|
|
|
|
|
if xml is None:
|
|
|
|
|
xml = build_sitemap(base)
|
|
|
|
|
_sitemap_cache.set(base, xml)
|
|
|
|
|
return Response(xml, media_type="application/xml")
|