feat: add seo diagnostics tool with cli commands, config paths, and static versioning

Add SEO_REPORTS_DIR to config, STATIC_VERSION for cache-busting, seo prune/clear CLI subcommands, tools router with seo job endpoints, and boot-versioned static URLs in Dockerfile and Makefile
This commit is contained in:
2026-06-14 00:16:22 +00:00
parent 61c1ae8c5d
commit 1b89f7c49f
110 changed files with 4501 additions and 270 deletions
+1
View File
@@ -26,6 +26,7 @@ CATEGORY_BY_PREFIX: dict[str, str] = {
"service": "service",
"container": "container",
"proxy": "ingress",
"seo": "tools",
"ai": "ai",
"devii": "devii",
"cli": "cli",
@@ -566,6 +566,35 @@ ACTIONS: tuple[Action, ...] = (
params=(path("uid", "Fork job uid returned by fork_project."),),
requires_auth=False,
),
Action(
name="seo_diagnostics",
method="POST",
path="/tools/seo/run",
summary="Run an SEO audit on a URL or sitemap",
description=(
"Queues a background SEO diagnostics job and returns {uid, status_url}. Poll the "
"status_url with seo_status until status is 'done', then share the score, grade and "
"report_url. mode='url' audits a single page; mode='sitemap' crawls up to max_pages."
),
params=(
body("url", "The page URL or sitemap.xml URL to audit.", required=True),
body("mode", "'url' for a single page or 'sitemap' to crawl a sitemap."),
body("max_pages", "Maximum pages to crawl in sitemap mode (1-50)."),
),
requires_auth=False,
),
Action(
name="seo_status",
method="GET",
path="/tools/seo/{uid}",
summary="Check an SEO audit and obtain its score once finished",
description=(
"Returns the audit status. When status is 'done', score, grade and report_url are "
"populated; while 'pending' or 'running', poll again shortly."
),
params=(path("uid", "SEO job uid returned by seo_diagnostics."),),
requires_auth=False,
),
Action(
name="search_users",
method="GET",
+4 -25
View File
@@ -13,6 +13,7 @@ from urllib.parse import urlparse
import httpx
from devplacepy.net_guard import effective_address, is_blocked_address
from ..config import Settings
from ..errors import NetworkError, ToolInputError, UpstreamError
from ..text import html_to_text
@@ -49,21 +50,6 @@ TITLE = re.compile(r"<title[^>]*>(.*?)</title>", re.IGNORECASE | re.DOTALL)
MIN_FETCH_CHARS = 1000
ALLOWED_METHODS = ("GET", "POST", "PUT", "PATCH", "DELETE", "HEAD", "OPTIONS")
NAT64_PREFIXES = (
ipaddress.ip_network("64:ff9b::/96"),
ipaddress.ip_network("64:ff9b:1::/48"),
)
def _effective_address(address: Any) -> Any:
if isinstance(address, ipaddress.IPv6Address):
if address.ipv4_mapped is not None:
return address.ipv4_mapped
for prefix in NAT64_PREFIXES:
if address in prefix:
return ipaddress.IPv4Address(int(address) & 0xFFFFFFFF)
return address
class FetchController:
def __init__(self, settings: Settings) -> None:
@@ -218,17 +204,10 @@ class FetchController:
except socket.gaierror as exc:
raise NetworkError(f"Could not resolve host: {host}", host=host) from exc
for info in infos:
address = _effective_address(ipaddress.ip_address(info[4][0]))
if (
address.is_private
or address.is_loopback
or address.is_link_local
or address.is_reserved
or address.is_multicast
or address.is_unspecified
):
address = ipaddress.ip_address(info[4][0])
if is_blocked_address(address):
raise ToolInputError(
f"Refusing to fetch a private or local address ({address}). "
f"Refusing to fetch a private or local address ({effective_address(address)}). "
"Set DEVII_FETCH_ALLOW_PRIVATE=1 to override."
)
+1
View File
@@ -0,0 +1 @@
# retoor <retoor@molodetz.nl>
@@ -0,0 +1 @@
# retoor <retoor@molodetz.nl>
@@ -0,0 +1,90 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import re
from .base import (
HIGH,
MEDIUM,
LOW,
INFO,
Check,
PageContext,
SiteContext,
ok_check,
page_check,
site_check,
)
CATEGORY = "ai_readiness"
_TAG = re.compile(r"<[^>]+>")
_SCRIPT_STYLE = re.compile(r"<(script|style)[^>]*>.*?</\1>", re.IGNORECASE | re.DOTALL)
def _text_words(html: str) -> int:
if not html:
return 0
stripped = _SCRIPT_STYLE.sub(" ", html)
stripped = _TAG.sub(" ", stripped)
return len([w for w in stripped.split() if w])
@page_check
def ssr_parity(page: PageContext, site: SiteContext) -> Check:
rendered = int(page.dom.get("wordCount", 0) or 0)
raw = _text_words(page.raw_html)
if rendered <= 0:
return None
ratio = raw / rendered if rendered else 0
passed = ratio >= 0.5
return ok_check(
"ai.ssr_parity",
CATEGORY,
"Content in initial HTML",
passed,
severity=HIGH,
value=f"{int(ratio * 100)}% server-rendered",
recommendation="Serve primary content in the initial HTML (SSR); JS-only content is invisible to many crawlers and AI agents.",
warn=ratio >= 0.25,
details={"raw_words": raw, "rendered_words": rendered},
url=page.url,
)
@page_check
def semantic_html(page: PageContext, site: SiteContext) -> Check:
semantic = page.dom.get("semantic", {}) or {}
has_main = int(semantic.get("main", 0) or 0) >= 1
has_landmarks = any(
int(semantic.get(tag, 0) or 0) >= 1 for tag in ("article", "nav", "header", "footer")
)
passed = has_main and has_landmarks
return ok_check(
"ai.semantic_html",
CATEGORY,
"Semantic HTML landmarks",
passed,
severity=LOW,
value="ok" if passed else "missing landmarks",
recommendation="Use <main>, <article>, <nav>, <header> and <footer> so machines can extract the main content.",
warn=True,
details=semantic,
url=page.url,
)
@site_check
def llms_txt(site: SiteContext) -> Check:
found = bool((site.llms_txt or {}).get("found"))
return ok_check(
"ai.llms_txt",
CATEGORY,
"llms.txt present",
found,
severity=LOW,
value="present" if found else "missing",
recommendation="Publish an /llms.txt manifest to guide AI crawlers to your key content (emerging standard).",
warn=True,
)
+128
View File
@@ -0,0 +1,128 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Callable, Optional
PASS = "pass"
WARN = "warn"
FAIL = "fail"
INFO = "info"
SKIP = "skip"
CRITICAL = "critical"
HIGH = "high"
MEDIUM = "medium"
LOW = "low"
INFORMATIONAL = "info"
SEVERITY_WEIGHT = {
CRITICAL: 5.0,
HIGH: 3.0,
MEDIUM: 2.0,
LOW: 1.0,
INFORMATIONAL: 0.0,
}
STATUS_CREDIT = {PASS: 1.0, WARN: 0.5, FAIL: 0.0, INFO: 1.0, SKIP: 1.0}
@dataclass
class Check:
id: str
category: str
title: str
status: str
severity: str = MEDIUM
value: str = ""
recommendation: str = ""
details: dict = field(default_factory=dict)
url: str = ""
def to_dict(self) -> dict:
return {
"id": self.id,
"category": self.category,
"title": self.title,
"status": self.status,
"severity": self.severity,
"value": self.value,
"recommendation": self.recommendation,
"details": self.details,
"url": self.url,
}
@dataclass
class PageContext:
requested_url: str = ""
url: str = ""
status: int = 0
ok: bool = False
error: str = ""
elapsed_ms: int = 0
redirect_chain: list = field(default_factory=list)
headers: dict = field(default_factory=dict)
raw_html: str = ""
rendered_html: str = ""
dom: dict = field(default_factory=dict)
metrics: dict = field(default_factory=dict)
mobile: dict = field(default_factory=dict)
screenshot: str = ""
@dataclass
class SiteContext:
target_url: str = ""
mode: str = "url"
base_host: str = ""
base_scheme: str = "https"
robots: dict = field(default_factory=dict)
sitemap: dict = field(default_factory=dict)
llms_txt: dict = field(default_factory=dict)
pages: list = field(default_factory=list)
PAGE_CHECKS: list[Callable] = []
SITE_CHECKS: list[Callable] = []
def page_check(func: Callable) -> Callable:
PAGE_CHECKS.append(func)
return func
def site_check(func: Callable) -> Callable:
SITE_CHECKS.append(func)
return func
def ok_check(
cid: str,
category: str,
title: str,
passed: bool,
*,
severity: str = MEDIUM,
value: str = "",
recommendation: str = "",
details: Optional[dict] = None,
url: str = "",
warn: bool = False,
) -> Check:
if passed:
status = PASS
else:
status = WARN if warn else FAIL
return Check(
id=cid,
category=category,
title=title,
status=status,
severity=severity,
value=value,
recommendation=recommendation if not passed else "",
details=details or {},
url=url,
)
@@ -0,0 +1,302 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from urllib.parse import urlparse
from .base import (
CRITICAL,
HIGH,
MEDIUM,
LOW,
PASS,
WARN,
FAIL,
INFO,
Check,
PageContext,
SiteContext,
ok_check,
page_check,
site_check,
)
CATEGORY = "crawlability"
def _normalize(url: str) -> str:
return url.rstrip("/").split("#")[0]
@page_check
def http_status(page: PageContext, site: SiteContext) -> Check:
passed = 200 <= page.status < 300
return ok_check(
"crawl.http_status",
CATEGORY,
"HTTP status",
passed,
severity=CRITICAL,
value=str(page.status),
recommendation="Return a 200 OK status for indexable pages.",
url=page.url,
)
@page_check
def redirects(page: PageContext, site: SiteContext) -> Check:
hops = len(page.redirect_chain)
if hops == 0:
return Check(
"crawl.redirects",
CATEGORY,
"Redirect chain",
PASS,
LOW,
value="no redirect",
url=page.url,
)
status = WARN if hops <= 2 else FAIL
return Check(
"crawl.redirects",
CATEGORY,
"Redirect chain",
status,
MEDIUM,
value=f"{hops} hop(s)",
recommendation="Link directly to the final URL to avoid redirect latency and link-equity loss.",
details={"chain": page.redirect_chain},
url=page.url,
)
@page_check
def https(page: PageContext, site: SiteContext) -> Check:
passed = page.url.startswith("https://")
return ok_check(
"crawl.https",
CATEGORY,
"Served over HTTPS",
passed,
severity=HIGH,
value="https" if passed else "http",
recommendation="Serve every page over HTTPS.",
url=page.url,
)
@page_check
def hsts(page: PageContext, site: SiteContext) -> Check:
if not page.url.startswith("https://"):
return None
present = bool(page.headers.get("strict-transport-security"))
return ok_check(
"crawl.hsts",
CATEGORY,
"HSTS header",
present,
severity=LOW,
value="present" if present else "missing",
recommendation="Add a Strict-Transport-Security header to enforce HTTPS.",
warn=True,
url=page.url,
)
@page_check
def canonical(page: PageContext, site: SiteContext) -> list:
href = page.dom.get("canonical", "")
count = int(page.dom.get("canonicalCount", 0) or 0)
results = [
ok_check(
"crawl.canonical_present",
CATEGORY,
"Canonical tag",
bool(href),
severity=MEDIUM,
value=href or "missing",
recommendation="Add a self-referential <link rel=canonical> to consolidate duplicates.",
warn=True,
url=page.url,
)
]
if count > 1:
results.append(
Check(
"crawl.canonical_single",
CATEGORY,
"Single canonical",
FAIL,
MEDIUM,
value=f"{count} canonicals",
recommendation="Declare exactly one canonical URL per page.",
url=page.url,
)
)
if href:
same = _normalize(href) == _normalize(page.url)
results.append(
ok_check(
"crawl.canonical_self",
CATEGORY,
"Self-referential canonical",
same,
severity=LOW,
value="self" if same else href,
recommendation="Point the canonical at this page's own URL unless intentionally consolidating.",
warn=True,
url=page.url,
)
)
return results
@page_check
def indexability(page: PageContext, site: SiteContext) -> Check:
robots = str(page.dom.get("metaRobots", "")).lower()
header = str(page.headers.get("x-robots-tag", "")).lower()
blocked = "noindex" in robots or "noindex" in header
return ok_check(
"crawl.indexable",
CATEGORY,
"Indexable (no noindex)",
not blocked,
severity=HIGH,
value="noindex" if blocked else "indexable",
recommendation="Remove noindex from meta robots / X-Robots-Tag if this page should rank.",
warn=True,
url=page.url,
)
@page_check
def url_hygiene(page: PageContext, site: SiteContext) -> Check:
parsed = urlparse(page.url)
path = parsed.path
issues = []
if len(page.url) > 100:
issues.append("long")
if any(c.isupper() for c in path):
issues.append("uppercase")
if "_" in path:
issues.append("underscores")
if parsed.query.count("&") >= 3:
issues.append("many params")
passed = not issues
return ok_check(
"crawl.url_hygiene",
CATEGORY,
"Clean URL",
passed,
severity=LOW,
value="clean" if passed else ", ".join(issues),
recommendation="Use short lowercase hyphenated paths with few query parameters.",
warn=True,
url=page.url,
)
@page_check
def mixed_content(page: PageContext, site: SiteContext) -> Check:
if not page.url.startswith("https://"):
return None
items = page.dom.get("mixedContent", []) or []
passed = not items
return ok_check(
"crawl.mixed_content",
CATEGORY,
"No mixed content",
passed,
severity=HIGH,
value="clean" if passed else f"{len(items)} insecure resource(s)",
recommendation="Load every sub-resource over HTTPS to avoid mixed-content blocking.",
details={"resources": items[:20]},
url=page.url,
)
@site_check
def robots_txt(site: SiteContext) -> list:
robots = site.robots or {}
fetched = robots.get("status") == 200
results = [
ok_check(
"crawl.robots_txt",
CATEGORY,
"robots.txt present",
fetched,
severity=MEDIUM,
value=f"{robots.get('status', 'n/a')}",
recommendation="Publish a robots.txt at the site root.",
warn=True,
)
]
if fetched:
results.append(
ok_check(
"crawl.robots_sitemap",
CATEGORY,
"Sitemap declared in robots.txt",
bool(robots.get("sitemap_urls")),
severity=LOW,
value=", ".join(robots.get("sitemap_urls", [])[:3]) or "none",
recommendation="Reference your sitemap with a Sitemap: directive in robots.txt.",
warn=True,
)
)
if robots.get("blocks_target"):
results.append(
Check(
"crawl.robots_blocks_target",
CATEGORY,
"robots.txt blocks the audited URL",
FAIL,
HIGH,
value="disallowed",
recommendation="The audited URL is disallowed in robots.txt; allow it if it should be crawled.",
)
)
return results
@site_check
def sitemap_xml(site: SiteContext) -> list:
sm = site.sitemap or {}
fetched = sm.get("status") == 200
results = [
ok_check(
"crawl.sitemap_present",
CATEGORY,
"XML sitemap reachable",
fetched,
severity=MEDIUM,
value=f"{sm.get('url_count', 0)} url(s)" if fetched else f"{sm.get('status', 'n/a')}",
recommendation="Publish an XML sitemap and submit it to search engines.",
warn=True,
)
]
if fetched:
results.append(
ok_check(
"crawl.sitemap_valid",
CATEGORY,
"Sitemap is valid XML",
bool(sm.get("valid_xml")),
severity=MEDIUM,
value="valid" if sm.get("valid_xml") else "invalid",
recommendation="Fix the sitemap XML so crawlers can parse it.",
)
)
results.append(
ok_check(
"crawl.sitemap_lastmod",
CATEGORY,
"Sitemap uses lastmod",
int(sm.get("lastmod_count", 0) or 0) > 0,
severity=LOW,
value=f"{sm.get('lastmod_count', 0)} entries",
recommendation="Add <lastmod> dates so crawlers prioritise fresh pages.",
warn=True,
)
)
return results
@@ -0,0 +1,79 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from collections import defaultdict
from .base import (
MEDIUM,
LOW,
INFO,
Check,
SiteContext,
site_check,
)
CATEGORY = "crosspage"
def _duplicates(site: SiteContext, dom_key: str) -> dict:
groups: dict[str, list] = defaultdict(list)
for page in site.pages:
value = (page.dom.get(dom_key, "") or "").strip().lower()
if value:
groups[value].append(page.url)
return {value: urls for value, urls in groups.items() if len(urls) > 1}
@site_check
def duplicate_titles(site: SiteContext) -> Check:
if len(site.pages) < 2:
return None
dupes = _duplicates(site, "title")
passed = not dupes
return Check(
"crosspage.duplicate_titles",
CATEGORY,
"Unique page titles",
"pass" if passed else "warn",
MEDIUM,
value="unique" if passed else f"{len(dupes)} duplicated title(s)",
recommendation="" if passed else "Give every page a distinct <title>.",
details={"duplicates": {k: v for k, v in list(dupes.items())[:10]}},
)
@site_check
def duplicate_descriptions(site: SiteContext) -> Check:
if len(site.pages) < 2:
return None
dupes = _duplicates(site, "metaDescription")
passed = not dupes
return Check(
"crosspage.duplicate_descriptions",
CATEGORY,
"Unique meta descriptions",
"pass" if passed else "warn",
LOW,
value="unique" if passed else f"{len(dupes)} duplicated description(s)",
recommendation="" if passed else "Write a distinct meta description for every page.",
details={"duplicates": {k: v for k, v in list(dupes.items())[:10]}},
)
@site_check
def hreflang_usage(site: SiteContext) -> Check:
annotated = [p for p in site.pages if p.dom.get("hreflang")]
if not annotated:
return None
return Check(
"crosspage.hreflang",
CATEGORY,
"Hreflang annotations",
INFO,
INFO,
value=f"{len(annotated)} page(s) declare hreflang",
details={
"pages": {p.url: p.dom.get("hreflang") for p in annotated[:5]},
},
)
@@ -0,0 +1,90 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from .base import (
MEDIUM,
LOW,
WARN,
Check,
PageContext,
SiteContext,
ok_check,
page_check,
)
CATEGORY = "content"
THIN_CONTENT_WORDS = 250
@page_check
def single_h1(page: PageContext, site: SiteContext) -> list:
h1s = page.dom.get("h1", []) or []
results = [
ok_check(
"content.h1_present",
CATEGORY,
"H1 heading",
len(h1s) >= 1,
severity=MEDIUM,
value=(h1s[0][:80] if h1s else "missing"),
recommendation="Add a single descriptive H1 heading.",
url=page.url,
)
]
if len(h1s) > 1:
results.append(
Check(
"content.h1_single",
CATEGORY,
"Single H1",
WARN,
LOW,
value=f"{len(h1s)} H1s",
recommendation="Use exactly one H1 per page.",
url=page.url,
)
)
return results
@page_check
def heading_order(page: PageContext, site: SiteContext) -> Check:
headings = page.dom.get("headings", []) or []
levels = [int(h.get("level", 0)) for h in headings if h.get("level")]
skipped = False
prev = 0
for level in levels:
if prev and level > prev + 1:
skipped = True
break
prev = level
return ok_check(
"content.heading_order",
CATEGORY,
"Heading hierarchy",
not skipped,
severity=LOW,
value="ordered" if not skipped else "skips a level",
recommendation="Do not skip heading levels (e.g. H2 to H4).",
warn=True,
url=page.url,
)
@page_check
def word_count(page: PageContext, site: SiteContext) -> Check:
words = int(page.dom.get("wordCount", 0) or 0)
passed = words >= THIN_CONTENT_WORDS
return ok_check(
"content.word_count",
CATEGORY,
"Content depth",
passed,
severity=MEDIUM,
value=f"{words} words",
recommendation=f"Thin content; aim for more than {THIN_CONTENT_WORDS} meaningful words.",
warn=True,
url=page.url,
)
@@ -0,0 +1,99 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from urllib.parse import urlparse
from .base import (
MEDIUM,
LOW,
INFO,
Check,
PageContext,
SiteContext,
ok_check,
page_check,
)
CATEGORY = "links"
GENERIC_ANCHORS = {
"click here",
"here",
"read more",
"more",
"link",
"this",
"this page",
}
@page_check
def link_counts(page: PageContext, site: SiteContext) -> Check:
links = page.dom.get("links", []) or []
host = site.base_host
internal = 0
external = 0
for link in links:
href = link.get("href", "")
netloc = urlparse(href).netloc
if not netloc or netloc == host:
internal += 1
else:
external += 1
return Check(
"links.counts",
CATEGORY,
"Link profile",
INFO,
INFO,
value=f"{internal} internal, {external} external",
details={"internal": internal, "external": external},
url=page.url,
)
@page_check
def internal_links_present(page: PageContext, site: SiteContext) -> Check:
links = page.dom.get("links", []) or []
host = site.base_host
internal = [
link
for link in links
if not urlparse(link.get("href", "")).netloc
or urlparse(link.get("href", "")).netloc == host
]
return ok_check(
"links.internal_present",
CATEGORY,
"Internal links",
len(internal) >= 1,
severity=LOW,
value=f"{len(internal)} internal link(s)",
recommendation="Add internal links so crawlers can discover related pages.",
warn=True,
url=page.url,
)
@page_check
def generic_anchors(page: PageContext, site: SiteContext) -> Check:
links = page.dom.get("links", []) or []
generic = [
link
for link in links
if (link.get("text", "") or "").strip().lower() in GENERIC_ANCHORS
]
passed = not generic
return ok_check(
"links.anchor_text",
CATEGORY,
"Descriptive anchor text",
passed,
severity=LOW,
value="descriptive" if passed else f"{len(generic)} generic anchor(s)",
recommendation="Replace generic anchors like 'click here' with descriptive text.",
warn=True,
details={"samples": [link.get("href", "") for link in generic[:10]]},
url=page.url,
)
+169
View File
@@ -0,0 +1,169 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from .base import (
HIGH,
MEDIUM,
LOW,
PASS,
WARN,
Check,
PageContext,
SiteContext,
ok_check,
page_check,
)
CATEGORY = "meta"
TITLE_MIN = 30
TITLE_MAX = 60
DESC_MIN = 70
DESC_MAX = 160
@page_check
def title(page: PageContext, site: SiteContext) -> list:
text = (page.dom.get("title", "") or "").strip()
count = int(page.dom.get("titleCount", 0) or 0)
results = [
ok_check(
"meta.title_present",
CATEGORY,
"Title tag",
bool(text),
severity=HIGH,
value=text or "missing",
recommendation="Add a unique, descriptive <title>.",
url=page.url,
)
]
if text:
length = len(text)
good = TITLE_MIN <= length <= TITLE_MAX
results.append(
ok_check(
"meta.title_length",
CATEGORY,
"Title length",
good,
severity=LOW,
value=f"{length} chars",
recommendation=f"Keep titles between {TITLE_MIN} and {TITLE_MAX} characters.",
warn=True,
url=page.url,
)
)
if count > 1:
results.append(
Check(
"meta.title_single",
CATEGORY,
"Single title tag",
WARN,
LOW,
value=f"{count} titles",
recommendation="Use exactly one <title> element.",
url=page.url,
)
)
return results
@page_check
def description(page: PageContext, site: SiteContext) -> list:
text = (page.dom.get("metaDescription", "") or "").strip()
results = [
ok_check(
"meta.description_present",
CATEGORY,
"Meta description",
bool(text),
severity=MEDIUM,
value=text[:120] or "missing",
recommendation="Write a compelling 70-160 character meta description.",
warn=True,
url=page.url,
)
]
if text:
length = len(text)
good = DESC_MIN <= length <= DESC_MAX
results.append(
ok_check(
"meta.description_length",
CATEGORY,
"Description length",
good,
severity=LOW,
value=f"{length} chars",
recommendation=f"Keep meta descriptions between {DESC_MIN} and {DESC_MAX} characters.",
warn=True,
url=page.url,
)
)
return results
@page_check
def lang(page: PageContext, site: SiteContext) -> Check:
value = (page.dom.get("htmlLang", "") or "").strip()
return ok_check(
"meta.html_lang",
CATEGORY,
"Document language",
bool(value),
severity=LOW,
value=value or "missing",
recommendation="Declare the page language with <html lang=...>.",
warn=True,
url=page.url,
)
@page_check
def charset(page: PageContext, site: SiteContext) -> Check:
value = (page.dom.get("charset", "") or "").strip()
return ok_check(
"meta.charset",
CATEGORY,
"Character encoding",
bool(value),
severity=LOW,
value=value or "missing",
recommendation="Declare a <meta charset> (UTF-8 recommended).",
warn=True,
url=page.url,
)
@page_check
def viewport(page: PageContext, site: SiteContext) -> Check:
present = bool(page.dom.get("hasViewport"))
return ok_check(
"meta.viewport",
CATEGORY,
"Mobile viewport",
present,
severity=HIGH,
value=page.dom.get("viewportContent", "") or ("present" if present else "missing"),
recommendation="Add <meta name=viewport content='width=device-width, initial-scale=1'>.",
url=page.url,
)
@page_check
def favicon(page: PageContext, site: SiteContext) -> Check:
present = bool(page.dom.get("favicon"))
return ok_check(
"meta.favicon",
CATEGORY,
"Favicon",
present,
severity=LOW,
value="present" if present else "missing",
recommendation="Provide a favicon for brand recognition in results and tabs.",
warn=True,
url=page.url,
)
@@ -0,0 +1,103 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from .base import (
HIGH,
MEDIUM,
LOW,
PageContext,
SiteContext,
ok_check,
page_check,
)
CATEGORY = "mobile_accessibility"
@page_check
def responsive(page: PageContext, site: SiteContext) -> PageContext:
mobile = page.mobile or {}
overflow = bool(mobile.get("hasHorizontalOverflow"))
return ok_check(
"mobile.no_overflow",
CATEGORY,
"No horizontal overflow (mobile)",
not overflow,
severity=MEDIUM,
value="ok" if not overflow else f"scrollWidth {mobile.get('scrollWidth', '?')}px",
recommendation="Eliminate horizontal scrolling at a 390px mobile viewport.",
warn=True,
url=page.url,
)
@page_check
def tap_targets(page: PageContext, site: SiteContext):
mobile = page.mobile or {}
small = int(mobile.get("smallTapTargets", 0) or 0)
if "smallTapTargets" not in mobile:
return None
return ok_check(
"mobile.tap_targets",
CATEGORY,
"Tap target sizing",
small == 0,
severity=LOW,
value="ok" if small == 0 else f"{small} small target(s)",
recommendation="Make interactive targets at least 24x24 px (48 px recommended).",
warn=True,
url=page.url,
)
@page_check
def image_alt(page: PageContext, site: SiteContext):
images = page.dom.get("images", []) or []
if not images:
return None
missing = [img for img in images if not (img.get("alt", "") or "").strip()]
return ok_check(
"a11y.image_alt",
CATEGORY,
"Image alt coverage",
not missing,
severity=MEDIUM,
value="complete" if not missing else f"{len(missing)}/{len(images)} missing alt",
recommendation="Add descriptive alt text to every meaningful image.",
warn=True,
url=page.url,
)
@page_check
def form_labels(page: PageContext, site: SiteContext):
missing = int(page.dom.get("formsMissingLabels", 0) or 0)
if not page.dom.get("formFieldCount"):
return None
return ok_check(
"a11y.form_labels",
CATEGORY,
"Form fields labelled",
missing == 0,
severity=LOW,
value="ok" if missing == 0 else f"{missing} unlabelled field(s)",
recommendation="Associate a <label> or aria-label with every form control.",
warn=True,
url=page.url,
)
@page_check
def document_title(page: PageContext, site: SiteContext):
return ok_check(
"a11y.document_title",
CATEGORY,
"Accessible document title",
bool((page.dom.get("title", "") or "").strip()),
severity=LOW,
value="present" if (page.dom.get("title", "") or "").strip() else "missing",
recommendation="Every page needs a non-empty <title> for assistive tech.",
warn=True,
url=page.url,
)
@@ -0,0 +1,283 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from .base import (
HIGH,
MEDIUM,
LOW,
INFO,
Check,
PageContext,
SiteContext,
ok_check,
page_check,
)
CATEGORY = "performance"
LCP_GOOD_MS = 2500
LCP_POOR_MS = 4000
CLS_GOOD = 0.1
CLS_POOR = 0.25
FCP_GOOD_MS = 1800
TTFB_GOOD_MS = 800
PAGE_WEIGHT_BUDGET = 3_000_000
REQUEST_BUDGET = 80
DOM_NODE_BUDGET = 1500
MODERN_FORMATS = (".webp", ".avif")
def _ms(metrics: dict, key: str) -> float:
try:
return float(metrics.get(key) or 0)
except (TypeError, ValueError):
return 0.0
@page_check
def lcp(page: PageContext, site: SiteContext) -> Check:
value = _ms(page.metrics, "lcp")
if value <= 0:
return None
good = value <= LCP_GOOD_MS
return ok_check(
"perf.lcp",
CATEGORY,
"Largest Contentful Paint",
good,
severity=HIGH,
value=f"{value / 1000:.2f}s",
recommendation=f"Reduce LCP below {LCP_GOOD_MS / 1000:.1f}s (optimise hero image/render path).",
warn=value <= LCP_POOR_MS,
url=page.url,
)
@page_check
def cls(page: PageContext, site: SiteContext) -> Check:
value = _ms(page.metrics, "cls")
good = value <= CLS_GOOD
return ok_check(
"perf.cls",
CATEGORY,
"Cumulative Layout Shift",
good,
severity=HIGH,
value=f"{value:.3f}",
recommendation=f"Keep CLS under {CLS_GOOD} by reserving space for media and ads.",
warn=value <= CLS_POOR,
url=page.url,
)
@page_check
def fcp(page: PageContext, site: SiteContext) -> Check:
value = _ms(page.metrics, "fcp")
if value <= 0:
return None
return ok_check(
"perf.fcp",
CATEGORY,
"First Contentful Paint",
value <= FCP_GOOD_MS,
severity=LOW,
value=f"{value / 1000:.2f}s",
recommendation=f"Reduce FCP below {FCP_GOOD_MS / 1000:.1f}s.",
warn=True,
url=page.url,
)
@page_check
def ttfb(page: PageContext, site: SiteContext) -> Check:
value = _ms(page.metrics, "ttfb")
if value <= 0:
return None
return ok_check(
"perf.ttfb",
CATEGORY,
"Time To First Byte",
value <= TTFB_GOOD_MS,
severity=MEDIUM,
value=f"{value:.0f}ms",
recommendation=f"Reduce server response time below {TTFB_GOOD_MS}ms.",
warn=True,
url=page.url,
)
@page_check
def page_weight(page: PageContext, site: SiteContext) -> Check:
total = int(page.metrics.get("transferSize") or 0)
if total <= 0:
return None
return ok_check(
"perf.page_weight",
CATEGORY,
"Total page weight",
total <= PAGE_WEIGHT_BUDGET,
severity=MEDIUM,
value=f"{total / 1_000_000:.2f} MB",
recommendation=f"Trim total transfer below {PAGE_WEIGHT_BUDGET / 1_000_000:.0f} MB.",
warn=True,
url=page.url,
)
@page_check
def request_count(page: PageContext, site: SiteContext) -> Check:
count = int(page.metrics.get("resourceCount") or 0)
if count <= 0:
return None
return ok_check(
"perf.requests",
CATEGORY,
"Request count",
count <= REQUEST_BUDGET,
severity=LOW,
value=str(count),
recommendation=f"Reduce HTTP requests below {REQUEST_BUDGET} (bundle, sprite, lazy-load).",
warn=True,
url=page.url,
)
@page_check
def dom_size(page: PageContext, site: SiteContext) -> Check:
nodes = int(page.metrics.get("domNodes") or 0)
if nodes <= 0:
return None
return ok_check(
"perf.dom_size",
CATEGORY,
"DOM size",
nodes <= DOM_NODE_BUDGET,
severity=LOW,
value=f"{nodes} nodes",
recommendation=f"Keep the DOM under {DOM_NODE_BUDGET} nodes for faster rendering.",
warn=True,
url=page.url,
)
@page_check
def compression(page: PageContext, site: SiteContext) -> Check:
encoding = str(page.headers.get("content-encoding", "")).lower()
passed = any(token in encoding for token in ("gzip", "br", "zstd", "deflate"))
return ok_check(
"perf.compression",
CATEGORY,
"Text compression",
passed,
severity=MEDIUM,
value=encoding or "none",
recommendation="Enable gzip or brotli compression on text responses.",
warn=True,
url=page.url,
)
@page_check
def caching(page: PageContext, site: SiteContext) -> Check:
cache = str(page.headers.get("cache-control", "")).lower()
passed = bool(cache) and "no-store" not in cache
return ok_check(
"perf.caching",
CATEGORY,
"Cache headers",
passed,
severity=LOW,
value=cache or "none",
recommendation="Set Cache-Control with sensible max-age for static assets.",
warn=True,
url=page.url,
)
@page_check
def http_version(page: PageContext, site: SiteContext) -> Check:
version = str(page.metrics.get("protocol", "") or "")
if not version:
return None
modern = any(token in version.lower() for token in ("h2", "h3", "http/2", "http/3"))
return ok_check(
"perf.http_version",
CATEGORY,
"Modern HTTP protocol",
modern,
severity=LOW,
value=version,
recommendation="Serve over HTTP/2 or HTTP/3 for multiplexed delivery.",
warn=True,
url=page.url,
)
@page_check
def console_errors(page: PageContext, site: SiteContext) -> Check:
errors = int(page.metrics.get("consoleErrors") or 0)
return ok_check(
"perf.console_errors",
CATEGORY,
"No console errors",
errors == 0,
severity=LOW,
value="clean" if errors == 0 else f"{errors} error(s)",
recommendation="Resolve JavaScript console errors emitted on load.",
warn=True,
details={"samples": page.metrics.get("consoleErrorSamples", [])[:5]},
url=page.url,
)
@page_check
def image_optimization(page: PageContext, site: SiteContext) -> list:
images = page.dom.get("images", []) or []
if not images:
return []
no_dimensions = [
img for img in images if not img.get("width") or not img.get("height")
]
legacy_format = [
img
for img in images
if (img.get("src", "") or "").lower().rsplit("?", 1)[0].endswith((".jpg", ".jpeg", ".png"))
]
not_lazy = [img for img in images if (img.get("loading", "") or "") != "lazy"]
results = [
ok_check(
"perf.img_dimensions",
CATEGORY,
"Images have dimensions",
not no_dimensions,
severity=MEDIUM,
value="ok" if not no_dimensions else f"{len(no_dimensions)} without width/height",
recommendation="Set width and height on images to prevent layout shift (CLS).",
warn=True,
url=page.url,
),
ok_check(
"perf.img_modern_format",
CATEGORY,
"Modern image formats",
not legacy_format,
severity=LOW,
value="ok" if not legacy_format else f"{len(legacy_format)} legacy image(s)",
recommendation="Serve WebP or AVIF instead of JPEG/PNG where possible.",
warn=True,
url=page.url,
),
ok_check(
"perf.img_lazy",
CATEGORY,
"Off-screen images lazy-loaded",
len(not_lazy) <= 1,
severity=LOW,
value="ok" if len(not_lazy) <= 1 else f"{len(not_lazy)} eager image(s)",
recommendation="Add loading=lazy to below-the-fold images.",
warn=True,
url=page.url,
),
]
return results
@@ -0,0 +1,131 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from .base import (
PAGE_CHECKS,
SITE_CHECKS,
SEVERITY_WEIGHT,
STATUS_CREDIT,
INFO,
SKIP,
Check,
PageContext,
SiteContext,
)
from . import crawl # noqa: F401
from . import meta # noqa: F401
from . import headings # noqa: F401
from . import links # noqa: F401
from . import structured_data # noqa: F401
from . import social # noqa: F401
from . import performance # noqa: F401
from . import mobile_a11y # noqa: F401
from . import security # noqa: F401
from . import ai_readiness # noqa: F401
from . import crosspage # noqa: F401
def _collect(result) -> list[Check]:
if result is None:
return []
if isinstance(result, Check):
return [result]
return [c for c in result if isinstance(c, Check)]
def run_page_checks(page: PageContext, site: SiteContext) -> list[Check]:
checks: list[Check] = []
for func in PAGE_CHECKS:
try:
checks.extend(_collect(func(page, site)))
except Exception as exc: # noqa: BLE001 - one bad check never aborts the page
checks.append(
Check(
id=f"{func.__name__}.error",
category="internal",
title=f"Check {func.__name__} failed",
status=INFO,
severity="info",
value=str(exc)[:200],
url=page.url,
)
)
return checks
def run_site_checks(site: SiteContext) -> list[Check]:
checks: list[Check] = []
for func in SITE_CHECKS:
try:
checks.extend(_collect(func(site)))
except Exception as exc: # noqa: BLE001
checks.append(
Check(
id=f"{func.__name__}.error",
category="internal",
title=f"Site check {func.__name__} failed",
status=INFO,
severity="info",
value=str(exc)[:200],
)
)
return checks
def compute_score(checks: list[Check]) -> dict:
by_category: dict[str, dict] = {}
earned = 0.0
weight = 0.0
counts = {"pass": 0, "warn": 0, "fail": 0, "info": 0, "skip": 0}
for check in checks:
counts[check.status] = counts.get(check.status, 0) + 1
bucket = by_category.setdefault(
check.category,
{"earned": 0.0, "weight": 0.0, "pass": 0, "warn": 0, "fail": 0, "info": 0, "skip": 0},
)
bucket[check.status] = bucket.get(check.status, 0) + 1
if check.status in (INFO, SKIP):
continue
w = SEVERITY_WEIGHT.get(check.severity, 1.0)
if w <= 0:
continue
credit = STATUS_CREDIT.get(check.status, 0.0)
earned += w * credit
weight += w
bucket["earned"] += w * credit
bucket["weight"] += w
categories = {}
for name, bucket in by_category.items():
cat_score = (
round(100 * bucket["earned"] / bucket["weight"])
if bucket["weight"] > 0
else None
)
categories[name] = {
"score": cat_score,
"pass": bucket["pass"],
"warn": bucket["warn"],
"fail": bucket["fail"],
"info": bucket["info"],
"skip": bucket["skip"],
}
score = round(100 * earned / weight) if weight > 0 else 0
return {
"score": score,
"grade": _grade(score),
"counts": counts,
"categories": categories,
}
def _grade(score: int) -> str:
if score >= 90:
return "A"
if score >= 80:
return "B"
if score >= 70:
return "C"
if score >= 55:
return "D"
return "F"
@@ -0,0 +1,59 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from .base import (
MEDIUM,
LOW,
PageContext,
SiteContext,
ok_check,
page_check,
)
CATEGORY = "security"
SECURITY_HEADERS = {
"content-security-policy": ("Content-Security-Policy", LOW),
"x-content-type-options": ("X-Content-Type-Options", LOW),
"x-frame-options": ("X-Frame-Options", LOW),
"referrer-policy": ("Referrer-Policy", LOW),
}
@page_check
def security_headers(page: PageContext, site: SiteContext) -> list:
results = []
for key, (label, severity) in SECURITY_HEADERS.items():
present = bool(page.headers.get(key))
results.append(
ok_check(
f"security.{key}",
CATEGORY,
label,
present,
severity=severity,
value="present" if present else "missing",
recommendation=f"Add the {label} response header.",
warn=True,
url=page.url,
)
)
return results
@page_check
def tls_valid(page: PageContext, site: SiteContext):
if "tlsValid" not in page.metrics:
return None
valid = bool(page.metrics.get("tlsValid"))
return ok_check(
"security.tls",
CATEGORY,
"Valid TLS certificate",
valid,
severity=MEDIUM,
value="valid" if valid else "invalid",
recommendation="Serve a valid, unexpired TLS certificate.",
url=page.url,
)
@@ -0,0 +1,59 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
from .base import (
MEDIUM,
LOW,
PageContext,
SiteContext,
ok_check,
page_check,
)
CATEGORY = "social"
OG_REQUIRED = ("og:title", "og:description", "og:image", "og:url")
TWITTER_REQUIRED = ("twitter:card",)
@page_check
def open_graph(page: PageContext, site: SiteContext) -> list:
og = page.dom.get("og", {}) or {}
missing = [tag for tag in OG_REQUIRED if not og.get(tag)]
results = [
ok_check(
"social.open_graph",
CATEGORY,
"Open Graph tags",
not missing,
severity=MEDIUM,
value="complete" if not missing else f"missing {', '.join(missing)}",
recommendation="Add og:title, og:description, og:image and og:url for rich social previews.",
warn=True,
details={"present": sorted(og.keys())},
url=page.url,
)
]
return results
@page_check
def twitter_card(page: PageContext, site: SiteContext) -> list:
twitter = page.dom.get("twitter", {}) or {}
og = page.dom.get("og", {}) or {}
has_card = bool(twitter.get("twitter:card"))
return [
ok_check(
"social.twitter_card",
CATEGORY,
"Twitter Card",
has_card or bool(og),
severity=LOW,
value=twitter.get("twitter:card", "") or ("falls back to OG" if og else "missing"),
recommendation="Add a twitter:card meta tag (summary_large_image) for X/Twitter previews.",
warn=True,
details={"present": sorted(twitter.keys())},
url=page.url,
)
]
@@ -0,0 +1,165 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import json
from .base import (
MEDIUM,
LOW,
INFO,
PASS,
WARN,
FAIL,
Check,
PageContext,
SiteContext,
ok_check,
page_check,
)
CATEGORY = "structured_data"
REQUIRED_PROPS = {
"Article": ("headline",),
"BlogPosting": ("headline",),
"NewsArticle": ("headline",),
"Product": ("name",),
"Offer": ("price", "priceCurrency"),
"Organization": ("name", "url"),
"LocalBusiness": ("name", "address"),
"BreadcrumbList": ("itemListElement",),
"FAQPage": ("mainEntity",),
"WebSite": ("name", "url"),
"Person": ("name",),
"Recipe": ("name", "recipeIngredient"),
"Event": ("name", "startDate"),
"VideoObject": ("name", "thumbnailUrl", "uploadDate"),
"SoftwareApplication": ("name",),
}
def _iter_objects(parsed):
if isinstance(parsed, list):
for item in parsed:
yield from _iter_objects(item)
return
if not isinstance(parsed, dict):
return
graph = parsed.get("@graph")
if isinstance(graph, list):
for item in graph:
yield from _iter_objects(item)
if parsed.get("@type"):
yield parsed
def _types(node) -> list:
raw = node.get("@type")
if isinstance(raw, list):
return [str(item) for item in raw]
return [str(raw)] if raw else []
@page_check
def jsonld(page: PageContext, site: SiteContext) -> list:
blocks = page.dom.get("jsonld", []) or []
present = bool(blocks)
results = [
ok_check(
"structured.jsonld_present",
CATEGORY,
"Structured data (JSON-LD)",
present,
severity=MEDIUM,
value=f"{len(blocks)} block(s)" if present else "none",
recommendation="Add JSON-LD structured data so search engines can render rich results.",
warn=True,
url=page.url,
)
]
if not present:
return results
invalid = 0
nodes = []
for block in blocks:
try:
parsed = json.loads(block)
except (ValueError, TypeError):
invalid += 1
continue
nodes.extend(_iter_objects(parsed))
results.append(
ok_check(
"structured.jsonld_valid",
CATEGORY,
"JSON-LD parses",
invalid == 0,
severity=MEDIUM,
value="valid" if invalid == 0 else f"{invalid} invalid block(s)",
recommendation="Fix JSON syntax errors in structured-data blocks.",
url=page.url,
)
)
found_types = sorted({t for node in nodes for t in _types(node)})
results.append(
Check(
"structured.types",
CATEGORY,
"Schema types",
INFO,
INFO,
value=", ".join(found_types) or "none",
details={"types": found_types},
url=page.url,
)
)
missing = []
for node in nodes:
for type_name in _types(node):
required = REQUIRED_PROPS.get(type_name)
if not required:
continue
absent = [prop for prop in required if not node.get(prop)]
if absent:
missing.append(f"{type_name}: {', '.join(absent)}")
if missing:
results.append(
Check(
"structured.required_props",
CATEGORY,
"Required schema properties",
WARN,
LOW,
value=f"{len(missing)} type(s) missing props",
recommendation="Add the required properties for each declared schema type.",
details={"missing": missing[:20]},
url=page.url,
)
)
return results
@page_check
def other_markup(page: PageContext, site: SiteContext) -> Check:
microdata = bool(page.dom.get("microdata"))
rdfa = bool(page.dom.get("rdfa"))
formats = []
if microdata:
formats.append("microdata")
if rdfa:
formats.append("RDFa")
return Check(
"structured.other_markup",
CATEGORY,
"Other structured-data formats",
INFO,
INFO,
value=", ".join(formats) or "none",
details={"microdata": microdata, "rdfa": rdfa},
url=page.url,
)
+422
View File
@@ -0,0 +1,422 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import asyncio
from urllib.parse import urljoin, urlparse
from xml.etree import ElementTree
import httpx
from devplacepy.net_guard import BlockedAddressError, guard_public_url
from .checks.base import PageContext, SiteContext
USER_AGENT = (
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/131.0.0.0 Safari/537.36 DevPlaceSEOBot/1.0"
)
NAV_TIMEOUT_MS = 30000
RAW_FETCH_TIMEOUT = 15.0
MAX_RAW_BYTES = 3_000_000
MOBILE_VIEWPORT = {"width": 390, "height": 844}
DESKTOP_VIEWPORT = {"width": 1366, "height": 900}
INIT_SCRIPT = """
window.__seo = { lcp: 0, cls: 0, consoleErrors: 0, errorSamples: [] };
try {
new PerformanceObserver((list) => {
for (const entry of list.getEntries()) {
window.__seo.lcp = Math.max(window.__seo.lcp, entry.startTime || entry.renderTime || 0);
}
}).observe({ type: 'largest-contentful-paint', buffered: true });
} catch (e) {}
try {
new PerformanceObserver((list) => {
for (const entry of list.getEntries()) {
if (!entry.hadRecentInput) window.__seo.cls += entry.value;
}
}).observe({ type: 'layout-shift', buffered: true });
} catch (e) {}
"""
EXTRACT_SCRIPT = r"""
() => {
const abs = (href) => { try { return new URL(href, document.baseURI).href; } catch (e) { return href || ''; } };
const metaByName = (name) => {
const el = document.querySelector(`meta[name="${name}"]`);
return el ? (el.getAttribute('content') || '') : '';
};
const og = {};
document.querySelectorAll('meta[property^="og:"]').forEach((m) => {
og[m.getAttribute('property')] = m.getAttribute('content') || '';
});
const twitter = {};
document.querySelectorAll('meta[name^="twitter:"]').forEach((m) => {
twitter[m.getAttribute('name')] = m.getAttribute('content') || '';
});
const headings = [];
document.querySelectorAll('h1,h2,h3,h4,h5,h6').forEach((h) => {
headings.push({ level: parseInt(h.tagName.substring(1), 10), text: (h.textContent || '').trim().slice(0, 160) });
});
const h1 = Array.from(document.querySelectorAll('h1')).map((h) => (h.textContent || '').trim().slice(0, 160));
const images = Array.from(document.querySelectorAll('img')).slice(0, 200).map((img) => ({
src: abs(img.getAttribute('src') || ''),
alt: img.getAttribute('alt'),
width: img.getAttribute('width') || (img.naturalWidth ? String(img.naturalWidth) : ''),
height: img.getAttribute('height') || (img.naturalHeight ? String(img.naturalHeight) : ''),
loading: img.getAttribute('loading') || ''
}));
const links = Array.from(document.querySelectorAll('a[href]')).slice(0, 500).map((a) => ({
href: abs(a.getAttribute('href') || ''),
rel: a.getAttribute('rel') || '',
text: (a.textContent || '').trim().slice(0, 120)
}));
const jsonld = Array.from(document.querySelectorAll('script[type="application/ld+json"]')).map((s) => s.textContent || '');
const hreflang = Array.from(document.querySelectorAll('link[rel="alternate"][hreflang]')).map((l) => ({
hreflang: l.getAttribute('hreflang'), href: abs(l.getAttribute('href') || '')
}));
const canonicalEls = document.querySelectorAll('link[rel="canonical"]');
const isHttps = location.protocol === 'https:';
const mixed = [];
if (isHttps) {
document.querySelectorAll('[src],[href]').forEach((el) => {
const v = el.getAttribute('src') || el.getAttribute('href') || '';
if (v.startsWith('http://')) mixed.push(v);
});
}
const viewportEl = document.querySelector('meta[name="viewport"]');
const charsetEl = document.querySelector('meta[charset]');
let formFieldCount = 0;
let formsMissingLabels = 0;
document.querySelectorAll('input,select,textarea').forEach((field) => {
const type = (field.getAttribute('type') || '').toLowerCase();
if (type === 'hidden' || type === 'submit' || type === 'button') return;
formFieldCount += 1;
const id = field.getAttribute('id');
const labelled = (id && document.querySelector(`label[for="${id}"]`)) ||
field.getAttribute('aria-label') || field.getAttribute('aria-labelledby') ||
field.closest('label');
if (!labelled) formsMissingLabels += 1;
});
const nav = performance.getEntriesByType('navigation')[0] || {};
const resources = performance.getEntriesByType('resource') || [];
let transfer = nav.transferSize || 0;
resources.forEach((r) => { transfer += (r.transferSize || 0); });
const bodyText = (document.body ? document.body.innerText || '' : '');
const wordCount = bodyText.split(/\s+/).filter(Boolean).length;
return {
title: (document.title || '').trim(),
titleCount: document.querySelectorAll('title').length,
metaDescription: metaByName('description'),
metaRobots: metaByName('robots'),
canonical: canonicalEls.length ? abs(canonicalEls[0].getAttribute('href') || '') : '',
canonicalCount: canonicalEls.length,
htmlLang: document.documentElement.getAttribute('lang') || '',
charset: charsetEl ? (charsetEl.getAttribute('charset') || '') : (document.characterSet || ''),
hasViewport: !!viewportEl,
viewportContent: viewportEl ? (viewportEl.getAttribute('content') || '') : '',
favicon: !!document.querySelector('link[rel~="icon"]'),
h1: h1,
headings: headings,
images: images,
links: links,
jsonld: jsonld,
hreflang: hreflang,
microdata: !!document.querySelector('[itemscope]'),
rdfa: !!document.querySelector('[vocab],[typeof],[property]'),
og: og,
twitter: twitter,
iframeCount: document.querySelectorAll('iframe').length,
semantic: {
main: document.querySelectorAll('main').length,
article: document.querySelectorAll('article').length,
nav: document.querySelectorAll('nav').length,
header: document.querySelectorAll('header').length,
footer: document.querySelectorAll('footer').length
},
mixedContent: mixed,
formFieldCount: formFieldCount,
formsMissingLabels: formsMissingLabels,
wordCount: wordCount,
metrics: {
ttfb: Math.max(0, (nav.responseStart || 0) - (nav.requestStart || 0)),
fcp: (performance.getEntriesByName('first-contentful-paint')[0] || {}).startTime || 0,
domContentLoaded: nav.domContentLoadedEventEnd || 0,
load: nav.loadEventEnd || 0,
transferSize: transfer,
resourceCount: resources.length,
domNodes: document.getElementsByTagName('*').length,
protocol: nav.nextHopProtocol || '',
lcp: window.__seo ? window.__seo.lcp : 0,
cls: window.__seo ? window.__seo.cls : 0
}
};
}
"""
MOBILE_SCRIPT = r"""
() => {
let small = 0;
document.querySelectorAll('a,button,[role="button"],input[type="submit"]').forEach((el) => {
const r = el.getBoundingClientRect();
if (r.width > 0 && r.height > 0 && (r.width < 24 || r.height < 24)) small += 1;
});
return {
scrollWidth: document.documentElement.scrollWidth,
clientWidth: document.documentElement.clientWidth,
hasHorizontalOverflow: document.documentElement.scrollWidth > document.documentElement.clientWidth + 4,
smallTapTargets: small
};
}
"""
def _normalize_url(url: str) -> str:
url = (url or "").strip()
if not url:
return ""
if "://" not in url:
url = "https://" + url
return url
async def _fetch_text(client: httpx.AsyncClient, url: str) -> dict:
try:
response = await client.get(url)
body = response.text[:MAX_RAW_BYTES]
return {"status": response.status_code, "text": body, "url": str(response.url)}
except httpx.HTTPError as exc:
return {"status": 0, "text": "", "url": url, "error": str(exc)[:200]}
def _parse_robots(text: str, target_path: str) -> dict:
disallows = []
sitemap_urls = []
active = False
for raw in text.splitlines():
line = raw.split("#", 1)[0].strip()
if not line:
continue
key, _, value = line.partition(":")
key = key.strip().lower()
value = value.strip()
if key == "user-agent":
active = value == "*"
elif key == "sitemap":
sitemap_urls.append(value)
elif key == "disallow" and active and value:
disallows.append(value)
blocks_target = any(
target_path.startswith(rule) for rule in disallows if rule and rule != "/"
) or "/" in [r for r in disallows]
return {
"disallows": disallows,
"sitemap_urls": sitemap_urls,
"blocks_target": blocks_target,
}
def _parse_sitemap(text: str) -> dict:
result = {"valid_xml": False, "url_count": 0, "lastmod_count": 0, "urls": []}
try:
root = ElementTree.fromstring(text)
except ElementTree.ParseError:
return result
result["valid_xml"] = True
urls = []
lastmods = 0
for url_el in root.iter():
tag = url_el.tag.rsplit("}", 1)[-1]
if tag == "loc" and url_el.text:
urls.append(url_el.text.strip())
elif tag == "lastmod" and url_el.text:
lastmods += 1
result["url_count"] = len(urls)
result["lastmod_count"] = lastmods
result["urls"] = urls
return result
async def _site_resources(
client: httpx.AsyncClient, base_url: str, sitemap_url: str, target_path: str
) -> tuple[dict, dict, dict]:
parsed = urlparse(base_url)
root = f"{parsed.scheme}://{parsed.netloc}"
robots_raw = await _fetch_text(client, urljoin(root + "/", "robots.txt"))
robots = dict(robots_raw)
if robots_raw.get("status") == 200:
robots.update(_parse_robots(robots_raw.get("text", ""), target_path))
sm_target = sitemap_url or urljoin(root + "/", "sitemap.xml")
sitemap_raw = await _fetch_text(client, sm_target)
sitemap = dict(sitemap_raw)
if sitemap_raw.get("status") == 200:
sitemap.update(_parse_sitemap(sitemap_raw.get("text", "")))
llms_raw = await _fetch_text(client, urljoin(root + "/", "llms.txt"))
llms = {"found": llms_raw.get("status") == 200, "status": llms_raw.get("status")}
return robots, sitemap, llms
async def _build_page(context, client, url: str, output_dir, index: int) -> PageContext:
page_ctx = PageContext(requested_url=url, url=url)
page = await context.new_page()
console_errors = []
def _on_console(message):
if message.type == "error":
console_errors.append(message.text[:200])
page.on("console", _on_console)
page.on("pageerror", lambda exc: console_errors.append(str(exc)[:200]))
await page.add_init_script(INIT_SCRIPT)
try:
response = await page.goto(url, wait_until="load", timeout=NAV_TIMEOUT_MS)
try:
await page.wait_for_load_state("networkidle", timeout=8000)
except Exception:
pass
page_ctx.status = response.status if response else 0
page_ctx.url = page.url
page_ctx.ok = bool(response and response.ok)
if response is not None:
page_ctx.headers = {k.lower(): v for k, v in response.headers.items()}
chain = []
req = response.request.redirected_from
while req is not None:
resp = await req.response()
chain.append({"url": req.url, "status": resp.status if resp else 0})
req = req.redirected_from
page_ctx.redirect_chain = list(reversed(chain))
dom = await page.evaluate(EXTRACT_SCRIPT)
page_ctx.metrics = dom.pop("metrics", {})
page_ctx.metrics["consoleErrors"] = len(console_errors)
page_ctx.metrics["consoleErrorSamples"] = console_errors[:5]
page_ctx.dom = dom
page_ctx.rendered_html = (await page.content())[:MAX_RAW_BYTES]
try:
await page.set_viewport_size(MOBILE_VIEWPORT)
page_ctx.mobile = await page.evaluate(MOBILE_SCRIPT)
await page.set_viewport_size(DESKTOP_VIEWPORT)
except Exception:
page_ctx.mobile = {}
if output_dir is not None:
shot = output_dir / f"page-{index}.png"
try:
await page.screenshot(path=str(shot), full_page=False)
page_ctx.screenshot = shot.name
except Exception:
page_ctx.screenshot = ""
raw = await _fetch_text(client, page_ctx.url)
page_ctx.raw_html = raw.get("text", "")
except Exception as exc: # noqa: BLE001 - record the failure as page state
page_ctx.error = str(exc)[:300]
page_ctx.ok = False
finally:
await page.close()
return page_ctx
async def crawl_target(payload: dict, emit, output_dir) -> SiteContext:
target = _normalize_url(payload.get("url", ""))
mode = payload.get("mode", "url")
allow_private = bool(payload.get("allow_private"))
max_pages = max(1, min(int(payload.get("max_pages", 1) or 1), 50))
await guard_public_url(target, allow_private=allow_private)
parsed = urlparse(target)
site = SiteContext(
target_url=target,
mode=mode,
base_host=parsed.netloc,
base_scheme=parsed.scheme,
)
from playwright.async_api import async_playwright
async with httpx.AsyncClient(
follow_redirects=True,
timeout=RAW_FETCH_TIMEOUT,
headers={"User-Agent": USER_AGENT},
) as client:
emit({"type": "stage", "stage": "resources", "message": "Reading robots.txt and sitemap"})
sitemap_hint = target if mode == "sitemap" else ""
robots, sitemap, llms = await _site_resources(
client, target, sitemap_hint, parsed.path or "/"
)
site.robots = robots
site.sitemap = sitemap
site.llms_txt = llms
if mode == "sitemap":
candidates = sitemap.get("urls", [])[:max_pages]
else:
candidates = [target]
safe_urls = []
for url in candidates:
host = urlparse(url).netloc
if host and host == site.base_host:
# Same host as the audited target, already validated by the
# top-level guard above; skip the redundant per-URL DNS lookup.
safe_urls.append(url)
continue
try:
await guard_public_url(url, allow_private=allow_private)
safe_urls.append(url)
except BlockedAddressError:
continue
if not safe_urls:
if mode == "sitemap":
raise ValueError(
"No crawlable page URLs found in the sitemap "
f"({sitemap.get('url_count', 0)} entries; none reachable or all blocked)."
)
safe_urls = [target]
emit(
{
"type": "target",
"url": target,
"mode": mode,
"urls": safe_urls,
"sitemap_total": sitemap.get("url_count", 0) if mode == "sitemap" else 0,
}
)
async with async_playwright() as pw:
browser = await pw.chromium.launch(
headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"]
)
context = await browser.new_context(
viewport=DESKTOP_VIEWPORT,
user_agent=USER_AGENT,
ignore_https_errors=False,
)
try:
total = len(safe_urls)
for index, url in enumerate(safe_urls):
emit(
{
"type": "progress",
"done": index,
"total": total,
"url": url,
"message": f"Auditing {url}",
}
)
page_ctx = await _build_page(
context, client, url, output_dir, index
)
site.pages.append(page_ctx)
yield_frame = {
"type": "page_loaded",
"url": page_ctx.url,
"status": page_ctx.status,
"done": index + 1,
"total": total,
}
emit(yield_frame)
finally:
await context.close()
await browser.close()
return site
+43
View File
@@ -0,0 +1,43 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import asyncio
MAX_BUFFER = 2000
class ProgressHub:
def __init__(self) -> None:
self._subscribers: dict[str, set[asyncio.Queue]] = {}
self._buffers: dict[str, list[dict]] = {}
def subscribe(self, uid: str) -> asyncio.Queue:
queue: asyncio.Queue = asyncio.Queue()
self._subscribers.setdefault(uid, set()).add(queue)
return queue
def unsubscribe(self, uid: str, queue: asyncio.Queue) -> None:
listeners = self._subscribers.get(uid)
if not listeners:
return
listeners.discard(queue)
if not listeners:
self._subscribers.pop(uid, None)
def publish(self, uid: str, frame: dict) -> None:
buffer = self._buffers.setdefault(uid, [])
buffer.append(frame)
if len(buffer) > MAX_BUFFER:
del buffer[: len(buffer) - MAX_BUFFER]
for queue in self._subscribers.get(uid, set()):
queue.put_nowait(frame)
def snapshot(self, uid: str) -> list[dict]:
return list(self._buffers.get(uid, []))
def clear(self, uid: str) -> None:
self._buffers.pop(uid, None)
hub = ProgressHub()
+155
View File
@@ -0,0 +1,155 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import asyncio
import json
import logging
import shutil
import sys
from pathlib import Path
from devplacepy.config import BASE_DIR, SEO_REPORTS_DIR
from devplacepy.services.jobs.base import JobService
from .progress import hub
logger = logging.getLogger(__name__)
WORKER_MODULE = "devplacepy.services.jobs.seo.worker"
STREAM_LIMIT = 16 * 1024 * 1024
class SeoService(JobService):
kind = "seo"
title = "SEO Diagnostics"
description = (
"Audits any URL or sitemap with a broad battery of technical, on-page, structured-data, "
"performance, accessibility and AI-readiness checks using a headless browser, streaming "
"live progress over a websocket and storing a categorised report."
)
def __init__(self):
super().__init__(name="seo", interval_seconds=2)
def report_dir(self, uid: str) -> Path:
return SEO_REPORTS_DIR / uid
async def process(self, job: dict) -> dict:
from devplacepy.services.audit import record as audit
uid = job["uid"]
payload = job.get("payload", {})
target = payload.get("url", "")
output_dir = self.report_dir(uid)
output_dir.mkdir(parents=True, exist_ok=True)
payload_path = output_dir / "payload.json"
payload_path.write_text(json.dumps(payload), encoding="utf-8")
actor_kind = "user" if job.get("owner_kind") == "user" else (
job.get("owner_kind") or "system"
)
actor_uid = job.get("owner_id") if job.get("owner_kind") == "user" else None
try:
summary = await self._run_worker(uid, payload_path, output_dir)
except Exception as exc:
hub.publish(uid, {"type": "failed", "message": str(exc)[:300]})
audit.record_system(
"seo.run.failed",
actor_kind=actor_kind,
actor_uid=actor_uid,
result="failure",
summary=f"SEO diagnostics for {target} failed",
metadata={"target": target, "error": str(exc)[:200]},
links=[audit.job(uid)],
)
raise
report = self._load_report(output_dir)
hub.publish(
uid,
{
"type": "done",
"score": summary.get("score"),
"grade": summary.get("grade"),
"page_count": summary.get("page_count"),
"report_url": f"/tools/seo/{uid}/report",
"report": report,
},
)
hub.clear(uid)
audit.record_system(
"seo.run.complete",
actor_kind=actor_kind,
actor_uid=actor_uid,
summary=f"SEO diagnostics for {target} scored {summary.get('score')}",
metadata={
"target": target,
"score": summary.get("score"),
"grade": summary.get("grade"),
"page_count": summary.get("page_count"),
},
links=[audit.job(uid)],
)
return {
"target": target,
"mode": payload.get("mode", "url"),
"score": summary.get("score", 0),
"grade": summary.get("grade", ""),
"page_count": summary.get("page_count", 0),
"counts": summary.get("counts", {}),
"categories": summary.get("categories", {}),
"report": report,
"report_url": f"/tools/seo/{uid}/report",
"bytes_in": 0,
"bytes_out": len(json.dumps(report)) if report else 0,
"item_count": summary.get("page_count", 0),
}
async def _run_worker(self, uid: str, payload_path: Path, output_dir: Path) -> dict:
proc = await asyncio.create_subprocess_exec(
sys.executable,
"-m",
WORKER_MODULE,
str(payload_path),
str(output_dir),
cwd=str(BASE_DIR),
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
limit=STREAM_LIMIT,
)
summary: dict = {}
worker_error = ""
while True:
line = await proc.stdout.readline()
if not line:
break
try:
frame = json.loads(line.decode("utf-8", "replace"))
except (ValueError, TypeError):
continue
hub.publish(uid, frame)
if frame.get("type") == "report_ready":
summary = frame
elif frame.get("type") == "error":
worker_error = frame.get("message", "worker error")
err = (await proc.stderr.read()).decode("utf-8", "replace")
await proc.wait()
if proc.returncode != 0 or not summary:
raise RuntimeError(
worker_error or err[:500] or f"seo worker exited {proc.returncode}"
)
return summary
def _load_report(self, output_dir: Path) -> dict:
report_path = output_dir / "report.json"
if not report_path.is_file():
return {}
try:
return json.loads(report_path.read_text(encoding="utf-8"))
except (ValueError, OSError):
return {}
def cleanup(self, job: dict) -> None:
hub.clear(job["uid"])
shutil.rmtree(self.report_dir(job["uid"]), ignore_errors=True)
+132
View File
@@ -0,0 +1,132 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import asyncio
import json
import sys
from datetime import datetime, timezone
from pathlib import Path
from .checks.registry import compute_score, run_page_checks, run_site_checks
from .crawler import crawl_target
def _emit(frame: dict) -> None:
sys.stdout.write(json.dumps(frame, ensure_ascii=False) + "\n")
sys.stdout.flush()
def _trim_site(site) -> dict:
robots = site.robots or {}
sitemap = site.sitemap or {}
return {
"robots": {
"status": robots.get("status"),
"disallows": robots.get("disallows", [])[:50],
"sitemap_urls": robots.get("sitemap_urls", [])[:10],
"blocks_target": robots.get("blocks_target", False),
},
"sitemap": {
"status": sitemap.get("status"),
"valid_xml": sitemap.get("valid_xml", False),
"url_count": sitemap.get("url_count", 0),
"lastmod_count": sitemap.get("lastmod_count", 0),
},
"llms_txt": site.llms_txt or {},
}
async def _run(payload: dict, output_dir: Path) -> dict:
site = await crawl_target(payload, _emit, output_dir)
all_checks = []
pages_summary = []
_emit({"type": "stage", "stage": "checks", "message": "Running SEO checks"})
for page in site.pages:
page_checks = run_page_checks(page, site)
all_checks.extend(page_checks)
page_score = compute_score(page_checks)
page_dicts = [c.to_dict() for c in page_checks]
pages_summary.append(
{
"url": page.url,
"requested_url": page.requested_url,
"status": page.status,
"ok": page.ok,
"error": page.error,
"screenshot": page.screenshot,
"score": page_score["score"],
"grade": page_score["grade"],
"counts": page_score["counts"],
}
)
_emit(
{
"type": "page",
"url": page.url,
"status": page.status,
"score": page_score["score"],
"grade": page_score["grade"],
"counts": page_score["counts"],
"checks": page_dicts,
}
)
site_checks = run_site_checks(site)
all_checks.extend(site_checks)
_emit(
{
"type": "site_checks",
"checks": [c.to_dict() for c in site_checks],
}
)
overall = compute_score(all_checks)
report = {
"target": site.target_url,
"mode": site.mode,
"generated_at": datetime.now(timezone.utc).isoformat(),
"page_count": len(site.pages),
"score": overall["score"],
"grade": overall["grade"],
"counts": overall["counts"],
"categories": overall["categories"],
"pages": pages_summary,
"checks": [c.to_dict() for c in all_checks],
"site": _trim_site(site),
}
if output_dir is not None:
(output_dir / "report.json").write_text(
json.dumps(report, ensure_ascii=False), encoding="utf-8"
)
_emit(
{
"type": "report_ready",
"score": report["score"],
"grade": report["grade"],
"counts": report["counts"],
"categories": report["categories"],
"page_count": report["page_count"],
}
)
return report
def main(argv: list) -> int:
if len(argv) != 3:
sys.stderr.write("usage: seo.worker <payload_json> <output_dir>\n")
return 2
payload = json.loads(Path(argv[1]).read_text(encoding="utf-8"))
output_dir = Path(argv[2])
output_dir.mkdir(parents=True, exist_ok=True)
try:
asyncio.run(_run(payload, output_dir))
except Exception as exc: # noqa: BLE001 - surface as a worker error line
_emit({"type": "error", "message": str(exc)[:500]})
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main(sys.argv))