forked from retoor/devplacepy
feat: add seo diagnostics tool with cli commands, config paths, and static versioning
Add SEO_REPORTS_DIR to config, STATIC_VERSION for cache-busting, seo prune/clear CLI subcommands, tools router with seo job endpoints, and boot-versioned static URLs in Dockerfile and Makefile
This commit is contained in:
@@ -26,6 +26,7 @@ CATEGORY_BY_PREFIX: dict[str, str] = {
|
||||
"service": "service",
|
||||
"container": "container",
|
||||
"proxy": "ingress",
|
||||
"seo": "tools",
|
||||
"ai": "ai",
|
||||
"devii": "devii",
|
||||
"cli": "cli",
|
||||
|
||||
@@ -566,6 +566,35 @@ ACTIONS: tuple[Action, ...] = (
|
||||
params=(path("uid", "Fork job uid returned by fork_project."),),
|
||||
requires_auth=False,
|
||||
),
|
||||
Action(
|
||||
name="seo_diagnostics",
|
||||
method="POST",
|
||||
path="/tools/seo/run",
|
||||
summary="Run an SEO audit on a URL or sitemap",
|
||||
description=(
|
||||
"Queues a background SEO diagnostics job and returns {uid, status_url}. Poll the "
|
||||
"status_url with seo_status until status is 'done', then share the score, grade and "
|
||||
"report_url. mode='url' audits a single page; mode='sitemap' crawls up to max_pages."
|
||||
),
|
||||
params=(
|
||||
body("url", "The page URL or sitemap.xml URL to audit.", required=True),
|
||||
body("mode", "'url' for a single page or 'sitemap' to crawl a sitemap."),
|
||||
body("max_pages", "Maximum pages to crawl in sitemap mode (1-50)."),
|
||||
),
|
||||
requires_auth=False,
|
||||
),
|
||||
Action(
|
||||
name="seo_status",
|
||||
method="GET",
|
||||
path="/tools/seo/{uid}",
|
||||
summary="Check an SEO audit and obtain its score once finished",
|
||||
description=(
|
||||
"Returns the audit status. When status is 'done', score, grade and report_url are "
|
||||
"populated; while 'pending' or 'running', poll again shortly."
|
||||
),
|
||||
params=(path("uid", "SEO job uid returned by seo_diagnostics."),),
|
||||
requires_auth=False,
|
||||
),
|
||||
Action(
|
||||
name="search_users",
|
||||
method="GET",
|
||||
|
||||
@@ -13,6 +13,7 @@ from urllib.parse import urlparse
|
||||
|
||||
import httpx
|
||||
|
||||
from devplacepy.net_guard import effective_address, is_blocked_address
|
||||
from ..config import Settings
|
||||
from ..errors import NetworkError, ToolInputError, UpstreamError
|
||||
from ..text import html_to_text
|
||||
@@ -49,21 +50,6 @@ TITLE = re.compile(r"<title[^>]*>(.*?)</title>", re.IGNORECASE | re.DOTALL)
|
||||
MIN_FETCH_CHARS = 1000
|
||||
ALLOWED_METHODS = ("GET", "POST", "PUT", "PATCH", "DELETE", "HEAD", "OPTIONS")
|
||||
|
||||
NAT64_PREFIXES = (
|
||||
ipaddress.ip_network("64:ff9b::/96"),
|
||||
ipaddress.ip_network("64:ff9b:1::/48"),
|
||||
)
|
||||
|
||||
|
||||
def _effective_address(address: Any) -> Any:
|
||||
if isinstance(address, ipaddress.IPv6Address):
|
||||
if address.ipv4_mapped is not None:
|
||||
return address.ipv4_mapped
|
||||
for prefix in NAT64_PREFIXES:
|
||||
if address in prefix:
|
||||
return ipaddress.IPv4Address(int(address) & 0xFFFFFFFF)
|
||||
return address
|
||||
|
||||
|
||||
class FetchController:
|
||||
def __init__(self, settings: Settings) -> None:
|
||||
@@ -218,17 +204,10 @@ class FetchController:
|
||||
except socket.gaierror as exc:
|
||||
raise NetworkError(f"Could not resolve host: {host}", host=host) from exc
|
||||
for info in infos:
|
||||
address = _effective_address(ipaddress.ip_address(info[4][0]))
|
||||
if (
|
||||
address.is_private
|
||||
or address.is_loopback
|
||||
or address.is_link_local
|
||||
or address.is_reserved
|
||||
or address.is_multicast
|
||||
or address.is_unspecified
|
||||
):
|
||||
address = ipaddress.ip_address(info[4][0])
|
||||
if is_blocked_address(address):
|
||||
raise ToolInputError(
|
||||
f"Refusing to fetch a private or local address ({address}). "
|
||||
f"Refusing to fetch a private or local address ({effective_address(address)}). "
|
||||
"Set DEVII_FETCH_ALLOW_PRIVATE=1 to override."
|
||||
)
|
||||
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
@@ -0,0 +1 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
@@ -0,0 +1,90 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from .base import (
|
||||
HIGH,
|
||||
MEDIUM,
|
||||
LOW,
|
||||
INFO,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
site_check,
|
||||
)
|
||||
|
||||
CATEGORY = "ai_readiness"
|
||||
|
||||
_TAG = re.compile(r"<[^>]+>")
|
||||
_SCRIPT_STYLE = re.compile(r"<(script|style)[^>]*>.*?</\1>", re.IGNORECASE | re.DOTALL)
|
||||
|
||||
|
||||
def _text_words(html: str) -> int:
|
||||
if not html:
|
||||
return 0
|
||||
stripped = _SCRIPT_STYLE.sub(" ", html)
|
||||
stripped = _TAG.sub(" ", stripped)
|
||||
return len([w for w in stripped.split() if w])
|
||||
|
||||
|
||||
@page_check
|
||||
def ssr_parity(page: PageContext, site: SiteContext) -> Check:
|
||||
rendered = int(page.dom.get("wordCount", 0) or 0)
|
||||
raw = _text_words(page.raw_html)
|
||||
if rendered <= 0:
|
||||
return None
|
||||
ratio = raw / rendered if rendered else 0
|
||||
passed = ratio >= 0.5
|
||||
return ok_check(
|
||||
"ai.ssr_parity",
|
||||
CATEGORY,
|
||||
"Content in initial HTML",
|
||||
passed,
|
||||
severity=HIGH,
|
||||
value=f"{int(ratio * 100)}% server-rendered",
|
||||
recommendation="Serve primary content in the initial HTML (SSR); JS-only content is invisible to many crawlers and AI agents.",
|
||||
warn=ratio >= 0.25,
|
||||
details={"raw_words": raw, "rendered_words": rendered},
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def semantic_html(page: PageContext, site: SiteContext) -> Check:
|
||||
semantic = page.dom.get("semantic", {}) or {}
|
||||
has_main = int(semantic.get("main", 0) or 0) >= 1
|
||||
has_landmarks = any(
|
||||
int(semantic.get(tag, 0) or 0) >= 1 for tag in ("article", "nav", "header", "footer")
|
||||
)
|
||||
passed = has_main and has_landmarks
|
||||
return ok_check(
|
||||
"ai.semantic_html",
|
||||
CATEGORY,
|
||||
"Semantic HTML landmarks",
|
||||
passed,
|
||||
severity=LOW,
|
||||
value="ok" if passed else "missing landmarks",
|
||||
recommendation="Use <main>, <article>, <nav>, <header> and <footer> so machines can extract the main content.",
|
||||
warn=True,
|
||||
details=semantic,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@site_check
|
||||
def llms_txt(site: SiteContext) -> Check:
|
||||
found = bool((site.llms_txt or {}).get("found"))
|
||||
return ok_check(
|
||||
"ai.llms_txt",
|
||||
CATEGORY,
|
||||
"llms.txt present",
|
||||
found,
|
||||
severity=LOW,
|
||||
value="present" if found else "missing",
|
||||
recommendation="Publish an /llms.txt manifest to guide AI crawlers to your key content (emerging standard).",
|
||||
warn=True,
|
||||
)
|
||||
@@ -0,0 +1,128 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Callable, Optional
|
||||
|
||||
PASS = "pass"
|
||||
WARN = "warn"
|
||||
FAIL = "fail"
|
||||
INFO = "info"
|
||||
SKIP = "skip"
|
||||
|
||||
CRITICAL = "critical"
|
||||
HIGH = "high"
|
||||
MEDIUM = "medium"
|
||||
LOW = "low"
|
||||
INFORMATIONAL = "info"
|
||||
|
||||
SEVERITY_WEIGHT = {
|
||||
CRITICAL: 5.0,
|
||||
HIGH: 3.0,
|
||||
MEDIUM: 2.0,
|
||||
LOW: 1.0,
|
||||
INFORMATIONAL: 0.0,
|
||||
}
|
||||
|
||||
STATUS_CREDIT = {PASS: 1.0, WARN: 0.5, FAIL: 0.0, INFO: 1.0, SKIP: 1.0}
|
||||
|
||||
|
||||
@dataclass
|
||||
class Check:
|
||||
id: str
|
||||
category: str
|
||||
title: str
|
||||
status: str
|
||||
severity: str = MEDIUM
|
||||
value: str = ""
|
||||
recommendation: str = ""
|
||||
details: dict = field(default_factory=dict)
|
||||
url: str = ""
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return {
|
||||
"id": self.id,
|
||||
"category": self.category,
|
||||
"title": self.title,
|
||||
"status": self.status,
|
||||
"severity": self.severity,
|
||||
"value": self.value,
|
||||
"recommendation": self.recommendation,
|
||||
"details": self.details,
|
||||
"url": self.url,
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class PageContext:
|
||||
requested_url: str = ""
|
||||
url: str = ""
|
||||
status: int = 0
|
||||
ok: bool = False
|
||||
error: str = ""
|
||||
elapsed_ms: int = 0
|
||||
redirect_chain: list = field(default_factory=list)
|
||||
headers: dict = field(default_factory=dict)
|
||||
raw_html: str = ""
|
||||
rendered_html: str = ""
|
||||
dom: dict = field(default_factory=dict)
|
||||
metrics: dict = field(default_factory=dict)
|
||||
mobile: dict = field(default_factory=dict)
|
||||
screenshot: str = ""
|
||||
|
||||
|
||||
@dataclass
|
||||
class SiteContext:
|
||||
target_url: str = ""
|
||||
mode: str = "url"
|
||||
base_host: str = ""
|
||||
base_scheme: str = "https"
|
||||
robots: dict = field(default_factory=dict)
|
||||
sitemap: dict = field(default_factory=dict)
|
||||
llms_txt: dict = field(default_factory=dict)
|
||||
pages: list = field(default_factory=list)
|
||||
|
||||
|
||||
PAGE_CHECKS: list[Callable] = []
|
||||
SITE_CHECKS: list[Callable] = []
|
||||
|
||||
|
||||
def page_check(func: Callable) -> Callable:
|
||||
PAGE_CHECKS.append(func)
|
||||
return func
|
||||
|
||||
|
||||
def site_check(func: Callable) -> Callable:
|
||||
SITE_CHECKS.append(func)
|
||||
return func
|
||||
|
||||
|
||||
def ok_check(
|
||||
cid: str,
|
||||
category: str,
|
||||
title: str,
|
||||
passed: bool,
|
||||
*,
|
||||
severity: str = MEDIUM,
|
||||
value: str = "",
|
||||
recommendation: str = "",
|
||||
details: Optional[dict] = None,
|
||||
url: str = "",
|
||||
warn: bool = False,
|
||||
) -> Check:
|
||||
if passed:
|
||||
status = PASS
|
||||
else:
|
||||
status = WARN if warn else FAIL
|
||||
return Check(
|
||||
id=cid,
|
||||
category=category,
|
||||
title=title,
|
||||
status=status,
|
||||
severity=severity,
|
||||
value=value,
|
||||
recommendation=recommendation if not passed else "",
|
||||
details=details or {},
|
||||
url=url,
|
||||
)
|
||||
@@ -0,0 +1,302 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from .base import (
|
||||
CRITICAL,
|
||||
HIGH,
|
||||
MEDIUM,
|
||||
LOW,
|
||||
PASS,
|
||||
WARN,
|
||||
FAIL,
|
||||
INFO,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
site_check,
|
||||
)
|
||||
|
||||
CATEGORY = "crawlability"
|
||||
|
||||
|
||||
def _normalize(url: str) -> str:
|
||||
return url.rstrip("/").split("#")[0]
|
||||
|
||||
|
||||
@page_check
|
||||
def http_status(page: PageContext, site: SiteContext) -> Check:
|
||||
passed = 200 <= page.status < 300
|
||||
return ok_check(
|
||||
"crawl.http_status",
|
||||
CATEGORY,
|
||||
"HTTP status",
|
||||
passed,
|
||||
severity=CRITICAL,
|
||||
value=str(page.status),
|
||||
recommendation="Return a 200 OK status for indexable pages.",
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def redirects(page: PageContext, site: SiteContext) -> Check:
|
||||
hops = len(page.redirect_chain)
|
||||
if hops == 0:
|
||||
return Check(
|
||||
"crawl.redirects",
|
||||
CATEGORY,
|
||||
"Redirect chain",
|
||||
PASS,
|
||||
LOW,
|
||||
value="no redirect",
|
||||
url=page.url,
|
||||
)
|
||||
status = WARN if hops <= 2 else FAIL
|
||||
return Check(
|
||||
"crawl.redirects",
|
||||
CATEGORY,
|
||||
"Redirect chain",
|
||||
status,
|
||||
MEDIUM,
|
||||
value=f"{hops} hop(s)",
|
||||
recommendation="Link directly to the final URL to avoid redirect latency and link-equity loss.",
|
||||
details={"chain": page.redirect_chain},
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def https(page: PageContext, site: SiteContext) -> Check:
|
||||
passed = page.url.startswith("https://")
|
||||
return ok_check(
|
||||
"crawl.https",
|
||||
CATEGORY,
|
||||
"Served over HTTPS",
|
||||
passed,
|
||||
severity=HIGH,
|
||||
value="https" if passed else "http",
|
||||
recommendation="Serve every page over HTTPS.",
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def hsts(page: PageContext, site: SiteContext) -> Check:
|
||||
if not page.url.startswith("https://"):
|
||||
return None
|
||||
present = bool(page.headers.get("strict-transport-security"))
|
||||
return ok_check(
|
||||
"crawl.hsts",
|
||||
CATEGORY,
|
||||
"HSTS header",
|
||||
present,
|
||||
severity=LOW,
|
||||
value="present" if present else "missing",
|
||||
recommendation="Add a Strict-Transport-Security header to enforce HTTPS.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def canonical(page: PageContext, site: SiteContext) -> list:
|
||||
href = page.dom.get("canonical", "")
|
||||
count = int(page.dom.get("canonicalCount", 0) or 0)
|
||||
results = [
|
||||
ok_check(
|
||||
"crawl.canonical_present",
|
||||
CATEGORY,
|
||||
"Canonical tag",
|
||||
bool(href),
|
||||
severity=MEDIUM,
|
||||
value=href or "missing",
|
||||
recommendation="Add a self-referential <link rel=canonical> to consolidate duplicates.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
]
|
||||
if count > 1:
|
||||
results.append(
|
||||
Check(
|
||||
"crawl.canonical_single",
|
||||
CATEGORY,
|
||||
"Single canonical",
|
||||
FAIL,
|
||||
MEDIUM,
|
||||
value=f"{count} canonicals",
|
||||
recommendation="Declare exactly one canonical URL per page.",
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
if href:
|
||||
same = _normalize(href) == _normalize(page.url)
|
||||
results.append(
|
||||
ok_check(
|
||||
"crawl.canonical_self",
|
||||
CATEGORY,
|
||||
"Self-referential canonical",
|
||||
same,
|
||||
severity=LOW,
|
||||
value="self" if same else href,
|
||||
recommendation="Point the canonical at this page's own URL unless intentionally consolidating.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
@page_check
|
||||
def indexability(page: PageContext, site: SiteContext) -> Check:
|
||||
robots = str(page.dom.get("metaRobots", "")).lower()
|
||||
header = str(page.headers.get("x-robots-tag", "")).lower()
|
||||
blocked = "noindex" in robots or "noindex" in header
|
||||
return ok_check(
|
||||
"crawl.indexable",
|
||||
CATEGORY,
|
||||
"Indexable (no noindex)",
|
||||
not blocked,
|
||||
severity=HIGH,
|
||||
value="noindex" if blocked else "indexable",
|
||||
recommendation="Remove noindex from meta robots / X-Robots-Tag if this page should rank.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def url_hygiene(page: PageContext, site: SiteContext) -> Check:
|
||||
parsed = urlparse(page.url)
|
||||
path = parsed.path
|
||||
issues = []
|
||||
if len(page.url) > 100:
|
||||
issues.append("long")
|
||||
if any(c.isupper() for c in path):
|
||||
issues.append("uppercase")
|
||||
if "_" in path:
|
||||
issues.append("underscores")
|
||||
if parsed.query.count("&") >= 3:
|
||||
issues.append("many params")
|
||||
passed = not issues
|
||||
return ok_check(
|
||||
"crawl.url_hygiene",
|
||||
CATEGORY,
|
||||
"Clean URL",
|
||||
passed,
|
||||
severity=LOW,
|
||||
value="clean" if passed else ", ".join(issues),
|
||||
recommendation="Use short lowercase hyphenated paths with few query parameters.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def mixed_content(page: PageContext, site: SiteContext) -> Check:
|
||||
if not page.url.startswith("https://"):
|
||||
return None
|
||||
items = page.dom.get("mixedContent", []) or []
|
||||
passed = not items
|
||||
return ok_check(
|
||||
"crawl.mixed_content",
|
||||
CATEGORY,
|
||||
"No mixed content",
|
||||
passed,
|
||||
severity=HIGH,
|
||||
value="clean" if passed else f"{len(items)} insecure resource(s)",
|
||||
recommendation="Load every sub-resource over HTTPS to avoid mixed-content blocking.",
|
||||
details={"resources": items[:20]},
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@site_check
|
||||
def robots_txt(site: SiteContext) -> list:
|
||||
robots = site.robots or {}
|
||||
fetched = robots.get("status") == 200
|
||||
results = [
|
||||
ok_check(
|
||||
"crawl.robots_txt",
|
||||
CATEGORY,
|
||||
"robots.txt present",
|
||||
fetched,
|
||||
severity=MEDIUM,
|
||||
value=f"{robots.get('status', 'n/a')}",
|
||||
recommendation="Publish a robots.txt at the site root.",
|
||||
warn=True,
|
||||
)
|
||||
]
|
||||
if fetched:
|
||||
results.append(
|
||||
ok_check(
|
||||
"crawl.robots_sitemap",
|
||||
CATEGORY,
|
||||
"Sitemap declared in robots.txt",
|
||||
bool(robots.get("sitemap_urls")),
|
||||
severity=LOW,
|
||||
value=", ".join(robots.get("sitemap_urls", [])[:3]) or "none",
|
||||
recommendation="Reference your sitemap with a Sitemap: directive in robots.txt.",
|
||||
warn=True,
|
||||
)
|
||||
)
|
||||
if robots.get("blocks_target"):
|
||||
results.append(
|
||||
Check(
|
||||
"crawl.robots_blocks_target",
|
||||
CATEGORY,
|
||||
"robots.txt blocks the audited URL",
|
||||
FAIL,
|
||||
HIGH,
|
||||
value="disallowed",
|
||||
recommendation="The audited URL is disallowed in robots.txt; allow it if it should be crawled.",
|
||||
)
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
@site_check
|
||||
def sitemap_xml(site: SiteContext) -> list:
|
||||
sm = site.sitemap or {}
|
||||
fetched = sm.get("status") == 200
|
||||
results = [
|
||||
ok_check(
|
||||
"crawl.sitemap_present",
|
||||
CATEGORY,
|
||||
"XML sitemap reachable",
|
||||
fetched,
|
||||
severity=MEDIUM,
|
||||
value=f"{sm.get('url_count', 0)} url(s)" if fetched else f"{sm.get('status', 'n/a')}",
|
||||
recommendation="Publish an XML sitemap and submit it to search engines.",
|
||||
warn=True,
|
||||
)
|
||||
]
|
||||
if fetched:
|
||||
results.append(
|
||||
ok_check(
|
||||
"crawl.sitemap_valid",
|
||||
CATEGORY,
|
||||
"Sitemap is valid XML",
|
||||
bool(sm.get("valid_xml")),
|
||||
severity=MEDIUM,
|
||||
value="valid" if sm.get("valid_xml") else "invalid",
|
||||
recommendation="Fix the sitemap XML so crawlers can parse it.",
|
||||
)
|
||||
)
|
||||
results.append(
|
||||
ok_check(
|
||||
"crawl.sitemap_lastmod",
|
||||
CATEGORY,
|
||||
"Sitemap uses lastmod",
|
||||
int(sm.get("lastmod_count", 0) or 0) > 0,
|
||||
severity=LOW,
|
||||
value=f"{sm.get('lastmod_count', 0)} entries",
|
||||
recommendation="Add <lastmod> dates so crawlers prioritise fresh pages.",
|
||||
warn=True,
|
||||
)
|
||||
)
|
||||
return results
|
||||
@@ -0,0 +1,79 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
|
||||
from .base import (
|
||||
MEDIUM,
|
||||
LOW,
|
||||
INFO,
|
||||
Check,
|
||||
SiteContext,
|
||||
site_check,
|
||||
)
|
||||
|
||||
CATEGORY = "crosspage"
|
||||
|
||||
|
||||
def _duplicates(site: SiteContext, dom_key: str) -> dict:
|
||||
groups: dict[str, list] = defaultdict(list)
|
||||
for page in site.pages:
|
||||
value = (page.dom.get(dom_key, "") or "").strip().lower()
|
||||
if value:
|
||||
groups[value].append(page.url)
|
||||
return {value: urls for value, urls in groups.items() if len(urls) > 1}
|
||||
|
||||
|
||||
@site_check
|
||||
def duplicate_titles(site: SiteContext) -> Check:
|
||||
if len(site.pages) < 2:
|
||||
return None
|
||||
dupes = _duplicates(site, "title")
|
||||
passed = not dupes
|
||||
return Check(
|
||||
"crosspage.duplicate_titles",
|
||||
CATEGORY,
|
||||
"Unique page titles",
|
||||
"pass" if passed else "warn",
|
||||
MEDIUM,
|
||||
value="unique" if passed else f"{len(dupes)} duplicated title(s)",
|
||||
recommendation="" if passed else "Give every page a distinct <title>.",
|
||||
details={"duplicates": {k: v for k, v in list(dupes.items())[:10]}},
|
||||
)
|
||||
|
||||
|
||||
@site_check
|
||||
def duplicate_descriptions(site: SiteContext) -> Check:
|
||||
if len(site.pages) < 2:
|
||||
return None
|
||||
dupes = _duplicates(site, "metaDescription")
|
||||
passed = not dupes
|
||||
return Check(
|
||||
"crosspage.duplicate_descriptions",
|
||||
CATEGORY,
|
||||
"Unique meta descriptions",
|
||||
"pass" if passed else "warn",
|
||||
LOW,
|
||||
value="unique" if passed else f"{len(dupes)} duplicated description(s)",
|
||||
recommendation="" if passed else "Write a distinct meta description for every page.",
|
||||
details={"duplicates": {k: v for k, v in list(dupes.items())[:10]}},
|
||||
)
|
||||
|
||||
|
||||
@site_check
|
||||
def hreflang_usage(site: SiteContext) -> Check:
|
||||
annotated = [p for p in site.pages if p.dom.get("hreflang")]
|
||||
if not annotated:
|
||||
return None
|
||||
return Check(
|
||||
"crosspage.hreflang",
|
||||
CATEGORY,
|
||||
"Hreflang annotations",
|
||||
INFO,
|
||||
INFO,
|
||||
value=f"{len(annotated)} page(s) declare hreflang",
|
||||
details={
|
||||
"pages": {p.url: p.dom.get("hreflang") for p in annotated[:5]},
|
||||
},
|
||||
)
|
||||
@@ -0,0 +1,90 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .base import (
|
||||
MEDIUM,
|
||||
LOW,
|
||||
WARN,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
)
|
||||
|
||||
CATEGORY = "content"
|
||||
|
||||
THIN_CONTENT_WORDS = 250
|
||||
|
||||
|
||||
@page_check
|
||||
def single_h1(page: PageContext, site: SiteContext) -> list:
|
||||
h1s = page.dom.get("h1", []) or []
|
||||
results = [
|
||||
ok_check(
|
||||
"content.h1_present",
|
||||
CATEGORY,
|
||||
"H1 heading",
|
||||
len(h1s) >= 1,
|
||||
severity=MEDIUM,
|
||||
value=(h1s[0][:80] if h1s else "missing"),
|
||||
recommendation="Add a single descriptive H1 heading.",
|
||||
url=page.url,
|
||||
)
|
||||
]
|
||||
if len(h1s) > 1:
|
||||
results.append(
|
||||
Check(
|
||||
"content.h1_single",
|
||||
CATEGORY,
|
||||
"Single H1",
|
||||
WARN,
|
||||
LOW,
|
||||
value=f"{len(h1s)} H1s",
|
||||
recommendation="Use exactly one H1 per page.",
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
@page_check
|
||||
def heading_order(page: PageContext, site: SiteContext) -> Check:
|
||||
headings = page.dom.get("headings", []) or []
|
||||
levels = [int(h.get("level", 0)) for h in headings if h.get("level")]
|
||||
skipped = False
|
||||
prev = 0
|
||||
for level in levels:
|
||||
if prev and level > prev + 1:
|
||||
skipped = True
|
||||
break
|
||||
prev = level
|
||||
return ok_check(
|
||||
"content.heading_order",
|
||||
CATEGORY,
|
||||
"Heading hierarchy",
|
||||
not skipped,
|
||||
severity=LOW,
|
||||
value="ordered" if not skipped else "skips a level",
|
||||
recommendation="Do not skip heading levels (e.g. H2 to H4).",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def word_count(page: PageContext, site: SiteContext) -> Check:
|
||||
words = int(page.dom.get("wordCount", 0) or 0)
|
||||
passed = words >= THIN_CONTENT_WORDS
|
||||
return ok_check(
|
||||
"content.word_count",
|
||||
CATEGORY,
|
||||
"Content depth",
|
||||
passed,
|
||||
severity=MEDIUM,
|
||||
value=f"{words} words",
|
||||
recommendation=f"Thin content; aim for more than {THIN_CONTENT_WORDS} meaningful words.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
@@ -0,0 +1,99 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from .base import (
|
||||
MEDIUM,
|
||||
LOW,
|
||||
INFO,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
)
|
||||
|
||||
CATEGORY = "links"
|
||||
|
||||
GENERIC_ANCHORS = {
|
||||
"click here",
|
||||
"here",
|
||||
"read more",
|
||||
"more",
|
||||
"link",
|
||||
"this",
|
||||
"this page",
|
||||
}
|
||||
|
||||
|
||||
@page_check
|
||||
def link_counts(page: PageContext, site: SiteContext) -> Check:
|
||||
links = page.dom.get("links", []) or []
|
||||
host = site.base_host
|
||||
internal = 0
|
||||
external = 0
|
||||
for link in links:
|
||||
href = link.get("href", "")
|
||||
netloc = urlparse(href).netloc
|
||||
if not netloc or netloc == host:
|
||||
internal += 1
|
||||
else:
|
||||
external += 1
|
||||
return Check(
|
||||
"links.counts",
|
||||
CATEGORY,
|
||||
"Link profile",
|
||||
INFO,
|
||||
INFO,
|
||||
value=f"{internal} internal, {external} external",
|
||||
details={"internal": internal, "external": external},
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def internal_links_present(page: PageContext, site: SiteContext) -> Check:
|
||||
links = page.dom.get("links", []) or []
|
||||
host = site.base_host
|
||||
internal = [
|
||||
link
|
||||
for link in links
|
||||
if not urlparse(link.get("href", "")).netloc
|
||||
or urlparse(link.get("href", "")).netloc == host
|
||||
]
|
||||
return ok_check(
|
||||
"links.internal_present",
|
||||
CATEGORY,
|
||||
"Internal links",
|
||||
len(internal) >= 1,
|
||||
severity=LOW,
|
||||
value=f"{len(internal)} internal link(s)",
|
||||
recommendation="Add internal links so crawlers can discover related pages.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def generic_anchors(page: PageContext, site: SiteContext) -> Check:
|
||||
links = page.dom.get("links", []) or []
|
||||
generic = [
|
||||
link
|
||||
for link in links
|
||||
if (link.get("text", "") or "").strip().lower() in GENERIC_ANCHORS
|
||||
]
|
||||
passed = not generic
|
||||
return ok_check(
|
||||
"links.anchor_text",
|
||||
CATEGORY,
|
||||
"Descriptive anchor text",
|
||||
passed,
|
||||
severity=LOW,
|
||||
value="descriptive" if passed else f"{len(generic)} generic anchor(s)",
|
||||
recommendation="Replace generic anchors like 'click here' with descriptive text.",
|
||||
warn=True,
|
||||
details={"samples": [link.get("href", "") for link in generic[:10]]},
|
||||
url=page.url,
|
||||
)
|
||||
@@ -0,0 +1,169 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .base import (
|
||||
HIGH,
|
||||
MEDIUM,
|
||||
LOW,
|
||||
PASS,
|
||||
WARN,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
)
|
||||
|
||||
CATEGORY = "meta"
|
||||
|
||||
TITLE_MIN = 30
|
||||
TITLE_MAX = 60
|
||||
DESC_MIN = 70
|
||||
DESC_MAX = 160
|
||||
|
||||
|
||||
@page_check
|
||||
def title(page: PageContext, site: SiteContext) -> list:
|
||||
text = (page.dom.get("title", "") or "").strip()
|
||||
count = int(page.dom.get("titleCount", 0) or 0)
|
||||
results = [
|
||||
ok_check(
|
||||
"meta.title_present",
|
||||
CATEGORY,
|
||||
"Title tag",
|
||||
bool(text),
|
||||
severity=HIGH,
|
||||
value=text or "missing",
|
||||
recommendation="Add a unique, descriptive <title>.",
|
||||
url=page.url,
|
||||
)
|
||||
]
|
||||
if text:
|
||||
length = len(text)
|
||||
good = TITLE_MIN <= length <= TITLE_MAX
|
||||
results.append(
|
||||
ok_check(
|
||||
"meta.title_length",
|
||||
CATEGORY,
|
||||
"Title length",
|
||||
good,
|
||||
severity=LOW,
|
||||
value=f"{length} chars",
|
||||
recommendation=f"Keep titles between {TITLE_MIN} and {TITLE_MAX} characters.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
if count > 1:
|
||||
results.append(
|
||||
Check(
|
||||
"meta.title_single",
|
||||
CATEGORY,
|
||||
"Single title tag",
|
||||
WARN,
|
||||
LOW,
|
||||
value=f"{count} titles",
|
||||
recommendation="Use exactly one <title> element.",
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
@page_check
|
||||
def description(page: PageContext, site: SiteContext) -> list:
|
||||
text = (page.dom.get("metaDescription", "") or "").strip()
|
||||
results = [
|
||||
ok_check(
|
||||
"meta.description_present",
|
||||
CATEGORY,
|
||||
"Meta description",
|
||||
bool(text),
|
||||
severity=MEDIUM,
|
||||
value=text[:120] or "missing",
|
||||
recommendation="Write a compelling 70-160 character meta description.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
]
|
||||
if text:
|
||||
length = len(text)
|
||||
good = DESC_MIN <= length <= DESC_MAX
|
||||
results.append(
|
||||
ok_check(
|
||||
"meta.description_length",
|
||||
CATEGORY,
|
||||
"Description length",
|
||||
good,
|
||||
severity=LOW,
|
||||
value=f"{length} chars",
|
||||
recommendation=f"Keep meta descriptions between {DESC_MIN} and {DESC_MAX} characters.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
@page_check
|
||||
def lang(page: PageContext, site: SiteContext) -> Check:
|
||||
value = (page.dom.get("htmlLang", "") or "").strip()
|
||||
return ok_check(
|
||||
"meta.html_lang",
|
||||
CATEGORY,
|
||||
"Document language",
|
||||
bool(value),
|
||||
severity=LOW,
|
||||
value=value or "missing",
|
||||
recommendation="Declare the page language with <html lang=...>.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def charset(page: PageContext, site: SiteContext) -> Check:
|
||||
value = (page.dom.get("charset", "") or "").strip()
|
||||
return ok_check(
|
||||
"meta.charset",
|
||||
CATEGORY,
|
||||
"Character encoding",
|
||||
bool(value),
|
||||
severity=LOW,
|
||||
value=value or "missing",
|
||||
recommendation="Declare a <meta charset> (UTF-8 recommended).",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def viewport(page: PageContext, site: SiteContext) -> Check:
|
||||
present = bool(page.dom.get("hasViewport"))
|
||||
return ok_check(
|
||||
"meta.viewport",
|
||||
CATEGORY,
|
||||
"Mobile viewport",
|
||||
present,
|
||||
severity=HIGH,
|
||||
value=page.dom.get("viewportContent", "") or ("present" if present else "missing"),
|
||||
recommendation="Add <meta name=viewport content='width=device-width, initial-scale=1'>.",
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def favicon(page: PageContext, site: SiteContext) -> Check:
|
||||
present = bool(page.dom.get("favicon"))
|
||||
return ok_check(
|
||||
"meta.favicon",
|
||||
CATEGORY,
|
||||
"Favicon",
|
||||
present,
|
||||
severity=LOW,
|
||||
value="present" if present else "missing",
|
||||
recommendation="Provide a favicon for brand recognition in results and tabs.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
@@ -0,0 +1,103 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .base import (
|
||||
HIGH,
|
||||
MEDIUM,
|
||||
LOW,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
)
|
||||
|
||||
CATEGORY = "mobile_accessibility"
|
||||
|
||||
|
||||
@page_check
|
||||
def responsive(page: PageContext, site: SiteContext) -> PageContext:
|
||||
mobile = page.mobile or {}
|
||||
overflow = bool(mobile.get("hasHorizontalOverflow"))
|
||||
return ok_check(
|
||||
"mobile.no_overflow",
|
||||
CATEGORY,
|
||||
"No horizontal overflow (mobile)",
|
||||
not overflow,
|
||||
severity=MEDIUM,
|
||||
value="ok" if not overflow else f"scrollWidth {mobile.get('scrollWidth', '?')}px",
|
||||
recommendation="Eliminate horizontal scrolling at a 390px mobile viewport.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def tap_targets(page: PageContext, site: SiteContext):
|
||||
mobile = page.mobile or {}
|
||||
small = int(mobile.get("smallTapTargets", 0) or 0)
|
||||
if "smallTapTargets" not in mobile:
|
||||
return None
|
||||
return ok_check(
|
||||
"mobile.tap_targets",
|
||||
CATEGORY,
|
||||
"Tap target sizing",
|
||||
small == 0,
|
||||
severity=LOW,
|
||||
value="ok" if small == 0 else f"{small} small target(s)",
|
||||
recommendation="Make interactive targets at least 24x24 px (48 px recommended).",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def image_alt(page: PageContext, site: SiteContext):
|
||||
images = page.dom.get("images", []) or []
|
||||
if not images:
|
||||
return None
|
||||
missing = [img for img in images if not (img.get("alt", "") or "").strip()]
|
||||
return ok_check(
|
||||
"a11y.image_alt",
|
||||
CATEGORY,
|
||||
"Image alt coverage",
|
||||
not missing,
|
||||
severity=MEDIUM,
|
||||
value="complete" if not missing else f"{len(missing)}/{len(images)} missing alt",
|
||||
recommendation="Add descriptive alt text to every meaningful image.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def form_labels(page: PageContext, site: SiteContext):
|
||||
missing = int(page.dom.get("formsMissingLabels", 0) or 0)
|
||||
if not page.dom.get("formFieldCount"):
|
||||
return None
|
||||
return ok_check(
|
||||
"a11y.form_labels",
|
||||
CATEGORY,
|
||||
"Form fields labelled",
|
||||
missing == 0,
|
||||
severity=LOW,
|
||||
value="ok" if missing == 0 else f"{missing} unlabelled field(s)",
|
||||
recommendation="Associate a <label> or aria-label with every form control.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def document_title(page: PageContext, site: SiteContext):
|
||||
return ok_check(
|
||||
"a11y.document_title",
|
||||
CATEGORY,
|
||||
"Accessible document title",
|
||||
bool((page.dom.get("title", "") or "").strip()),
|
||||
severity=LOW,
|
||||
value="present" if (page.dom.get("title", "") or "").strip() else "missing",
|
||||
recommendation="Every page needs a non-empty <title> for assistive tech.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
@@ -0,0 +1,283 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .base import (
|
||||
HIGH,
|
||||
MEDIUM,
|
||||
LOW,
|
||||
INFO,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
)
|
||||
|
||||
CATEGORY = "performance"
|
||||
|
||||
LCP_GOOD_MS = 2500
|
||||
LCP_POOR_MS = 4000
|
||||
CLS_GOOD = 0.1
|
||||
CLS_POOR = 0.25
|
||||
FCP_GOOD_MS = 1800
|
||||
TTFB_GOOD_MS = 800
|
||||
PAGE_WEIGHT_BUDGET = 3_000_000
|
||||
REQUEST_BUDGET = 80
|
||||
DOM_NODE_BUDGET = 1500
|
||||
MODERN_FORMATS = (".webp", ".avif")
|
||||
|
||||
|
||||
def _ms(metrics: dict, key: str) -> float:
|
||||
try:
|
||||
return float(metrics.get(key) or 0)
|
||||
except (TypeError, ValueError):
|
||||
return 0.0
|
||||
|
||||
|
||||
@page_check
|
||||
def lcp(page: PageContext, site: SiteContext) -> Check:
|
||||
value = _ms(page.metrics, "lcp")
|
||||
if value <= 0:
|
||||
return None
|
||||
good = value <= LCP_GOOD_MS
|
||||
return ok_check(
|
||||
"perf.lcp",
|
||||
CATEGORY,
|
||||
"Largest Contentful Paint",
|
||||
good,
|
||||
severity=HIGH,
|
||||
value=f"{value / 1000:.2f}s",
|
||||
recommendation=f"Reduce LCP below {LCP_GOOD_MS / 1000:.1f}s (optimise hero image/render path).",
|
||||
warn=value <= LCP_POOR_MS,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def cls(page: PageContext, site: SiteContext) -> Check:
|
||||
value = _ms(page.metrics, "cls")
|
||||
good = value <= CLS_GOOD
|
||||
return ok_check(
|
||||
"perf.cls",
|
||||
CATEGORY,
|
||||
"Cumulative Layout Shift",
|
||||
good,
|
||||
severity=HIGH,
|
||||
value=f"{value:.3f}",
|
||||
recommendation=f"Keep CLS under {CLS_GOOD} by reserving space for media and ads.",
|
||||
warn=value <= CLS_POOR,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def fcp(page: PageContext, site: SiteContext) -> Check:
|
||||
value = _ms(page.metrics, "fcp")
|
||||
if value <= 0:
|
||||
return None
|
||||
return ok_check(
|
||||
"perf.fcp",
|
||||
CATEGORY,
|
||||
"First Contentful Paint",
|
||||
value <= FCP_GOOD_MS,
|
||||
severity=LOW,
|
||||
value=f"{value / 1000:.2f}s",
|
||||
recommendation=f"Reduce FCP below {FCP_GOOD_MS / 1000:.1f}s.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def ttfb(page: PageContext, site: SiteContext) -> Check:
|
||||
value = _ms(page.metrics, "ttfb")
|
||||
if value <= 0:
|
||||
return None
|
||||
return ok_check(
|
||||
"perf.ttfb",
|
||||
CATEGORY,
|
||||
"Time To First Byte",
|
||||
value <= TTFB_GOOD_MS,
|
||||
severity=MEDIUM,
|
||||
value=f"{value:.0f}ms",
|
||||
recommendation=f"Reduce server response time below {TTFB_GOOD_MS}ms.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def page_weight(page: PageContext, site: SiteContext) -> Check:
|
||||
total = int(page.metrics.get("transferSize") or 0)
|
||||
if total <= 0:
|
||||
return None
|
||||
return ok_check(
|
||||
"perf.page_weight",
|
||||
CATEGORY,
|
||||
"Total page weight",
|
||||
total <= PAGE_WEIGHT_BUDGET,
|
||||
severity=MEDIUM,
|
||||
value=f"{total / 1_000_000:.2f} MB",
|
||||
recommendation=f"Trim total transfer below {PAGE_WEIGHT_BUDGET / 1_000_000:.0f} MB.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def request_count(page: PageContext, site: SiteContext) -> Check:
|
||||
count = int(page.metrics.get("resourceCount") or 0)
|
||||
if count <= 0:
|
||||
return None
|
||||
return ok_check(
|
||||
"perf.requests",
|
||||
CATEGORY,
|
||||
"Request count",
|
||||
count <= REQUEST_BUDGET,
|
||||
severity=LOW,
|
||||
value=str(count),
|
||||
recommendation=f"Reduce HTTP requests below {REQUEST_BUDGET} (bundle, sprite, lazy-load).",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def dom_size(page: PageContext, site: SiteContext) -> Check:
|
||||
nodes = int(page.metrics.get("domNodes") or 0)
|
||||
if nodes <= 0:
|
||||
return None
|
||||
return ok_check(
|
||||
"perf.dom_size",
|
||||
CATEGORY,
|
||||
"DOM size",
|
||||
nodes <= DOM_NODE_BUDGET,
|
||||
severity=LOW,
|
||||
value=f"{nodes} nodes",
|
||||
recommendation=f"Keep the DOM under {DOM_NODE_BUDGET} nodes for faster rendering.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def compression(page: PageContext, site: SiteContext) -> Check:
|
||||
encoding = str(page.headers.get("content-encoding", "")).lower()
|
||||
passed = any(token in encoding for token in ("gzip", "br", "zstd", "deflate"))
|
||||
return ok_check(
|
||||
"perf.compression",
|
||||
CATEGORY,
|
||||
"Text compression",
|
||||
passed,
|
||||
severity=MEDIUM,
|
||||
value=encoding or "none",
|
||||
recommendation="Enable gzip or brotli compression on text responses.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def caching(page: PageContext, site: SiteContext) -> Check:
|
||||
cache = str(page.headers.get("cache-control", "")).lower()
|
||||
passed = bool(cache) and "no-store" not in cache
|
||||
return ok_check(
|
||||
"perf.caching",
|
||||
CATEGORY,
|
||||
"Cache headers",
|
||||
passed,
|
||||
severity=LOW,
|
||||
value=cache or "none",
|
||||
recommendation="Set Cache-Control with sensible max-age for static assets.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def http_version(page: PageContext, site: SiteContext) -> Check:
|
||||
version = str(page.metrics.get("protocol", "") or "")
|
||||
if not version:
|
||||
return None
|
||||
modern = any(token in version.lower() for token in ("h2", "h3", "http/2", "http/3"))
|
||||
return ok_check(
|
||||
"perf.http_version",
|
||||
CATEGORY,
|
||||
"Modern HTTP protocol",
|
||||
modern,
|
||||
severity=LOW,
|
||||
value=version,
|
||||
recommendation="Serve over HTTP/2 or HTTP/3 for multiplexed delivery.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def console_errors(page: PageContext, site: SiteContext) -> Check:
|
||||
errors = int(page.metrics.get("consoleErrors") or 0)
|
||||
return ok_check(
|
||||
"perf.console_errors",
|
||||
CATEGORY,
|
||||
"No console errors",
|
||||
errors == 0,
|
||||
severity=LOW,
|
||||
value="clean" if errors == 0 else f"{errors} error(s)",
|
||||
recommendation="Resolve JavaScript console errors emitted on load.",
|
||||
warn=True,
|
||||
details={"samples": page.metrics.get("consoleErrorSamples", [])[:5]},
|
||||
url=page.url,
|
||||
)
|
||||
|
||||
|
||||
@page_check
|
||||
def image_optimization(page: PageContext, site: SiteContext) -> list:
|
||||
images = page.dom.get("images", []) or []
|
||||
if not images:
|
||||
return []
|
||||
no_dimensions = [
|
||||
img for img in images if not img.get("width") or not img.get("height")
|
||||
]
|
||||
legacy_format = [
|
||||
img
|
||||
for img in images
|
||||
if (img.get("src", "") or "").lower().rsplit("?", 1)[0].endswith((".jpg", ".jpeg", ".png"))
|
||||
]
|
||||
not_lazy = [img for img in images if (img.get("loading", "") or "") != "lazy"]
|
||||
results = [
|
||||
ok_check(
|
||||
"perf.img_dimensions",
|
||||
CATEGORY,
|
||||
"Images have dimensions",
|
||||
not no_dimensions,
|
||||
severity=MEDIUM,
|
||||
value="ok" if not no_dimensions else f"{len(no_dimensions)} without width/height",
|
||||
recommendation="Set width and height on images to prevent layout shift (CLS).",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
),
|
||||
ok_check(
|
||||
"perf.img_modern_format",
|
||||
CATEGORY,
|
||||
"Modern image formats",
|
||||
not legacy_format,
|
||||
severity=LOW,
|
||||
value="ok" if not legacy_format else f"{len(legacy_format)} legacy image(s)",
|
||||
recommendation="Serve WebP or AVIF instead of JPEG/PNG where possible.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
),
|
||||
ok_check(
|
||||
"perf.img_lazy",
|
||||
CATEGORY,
|
||||
"Off-screen images lazy-loaded",
|
||||
len(not_lazy) <= 1,
|
||||
severity=LOW,
|
||||
value="ok" if len(not_lazy) <= 1 else f"{len(not_lazy)} eager image(s)",
|
||||
recommendation="Add loading=lazy to below-the-fold images.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
),
|
||||
]
|
||||
return results
|
||||
@@ -0,0 +1,131 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .base import (
|
||||
PAGE_CHECKS,
|
||||
SITE_CHECKS,
|
||||
SEVERITY_WEIGHT,
|
||||
STATUS_CREDIT,
|
||||
INFO,
|
||||
SKIP,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
)
|
||||
from . import crawl # noqa: F401
|
||||
from . import meta # noqa: F401
|
||||
from . import headings # noqa: F401
|
||||
from . import links # noqa: F401
|
||||
from . import structured_data # noqa: F401
|
||||
from . import social # noqa: F401
|
||||
from . import performance # noqa: F401
|
||||
from . import mobile_a11y # noqa: F401
|
||||
from . import security # noqa: F401
|
||||
from . import ai_readiness # noqa: F401
|
||||
from . import crosspage # noqa: F401
|
||||
|
||||
|
||||
def _collect(result) -> list[Check]:
|
||||
if result is None:
|
||||
return []
|
||||
if isinstance(result, Check):
|
||||
return [result]
|
||||
return [c for c in result if isinstance(c, Check)]
|
||||
|
||||
|
||||
def run_page_checks(page: PageContext, site: SiteContext) -> list[Check]:
|
||||
checks: list[Check] = []
|
||||
for func in PAGE_CHECKS:
|
||||
try:
|
||||
checks.extend(_collect(func(page, site)))
|
||||
except Exception as exc: # noqa: BLE001 - one bad check never aborts the page
|
||||
checks.append(
|
||||
Check(
|
||||
id=f"{func.__name__}.error",
|
||||
category="internal",
|
||||
title=f"Check {func.__name__} failed",
|
||||
status=INFO,
|
||||
severity="info",
|
||||
value=str(exc)[:200],
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
return checks
|
||||
|
||||
|
||||
def run_site_checks(site: SiteContext) -> list[Check]:
|
||||
checks: list[Check] = []
|
||||
for func in SITE_CHECKS:
|
||||
try:
|
||||
checks.extend(_collect(func(site)))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
checks.append(
|
||||
Check(
|
||||
id=f"{func.__name__}.error",
|
||||
category="internal",
|
||||
title=f"Site check {func.__name__} failed",
|
||||
status=INFO,
|
||||
severity="info",
|
||||
value=str(exc)[:200],
|
||||
)
|
||||
)
|
||||
return checks
|
||||
|
||||
|
||||
def compute_score(checks: list[Check]) -> dict:
|
||||
by_category: dict[str, dict] = {}
|
||||
earned = 0.0
|
||||
weight = 0.0
|
||||
counts = {"pass": 0, "warn": 0, "fail": 0, "info": 0, "skip": 0}
|
||||
for check in checks:
|
||||
counts[check.status] = counts.get(check.status, 0) + 1
|
||||
bucket = by_category.setdefault(
|
||||
check.category,
|
||||
{"earned": 0.0, "weight": 0.0, "pass": 0, "warn": 0, "fail": 0, "info": 0, "skip": 0},
|
||||
)
|
||||
bucket[check.status] = bucket.get(check.status, 0) + 1
|
||||
if check.status in (INFO, SKIP):
|
||||
continue
|
||||
w = SEVERITY_WEIGHT.get(check.severity, 1.0)
|
||||
if w <= 0:
|
||||
continue
|
||||
credit = STATUS_CREDIT.get(check.status, 0.0)
|
||||
earned += w * credit
|
||||
weight += w
|
||||
bucket["earned"] += w * credit
|
||||
bucket["weight"] += w
|
||||
categories = {}
|
||||
for name, bucket in by_category.items():
|
||||
cat_score = (
|
||||
round(100 * bucket["earned"] / bucket["weight"])
|
||||
if bucket["weight"] > 0
|
||||
else None
|
||||
)
|
||||
categories[name] = {
|
||||
"score": cat_score,
|
||||
"pass": bucket["pass"],
|
||||
"warn": bucket["warn"],
|
||||
"fail": bucket["fail"],
|
||||
"info": bucket["info"],
|
||||
"skip": bucket["skip"],
|
||||
}
|
||||
score = round(100 * earned / weight) if weight > 0 else 0
|
||||
return {
|
||||
"score": score,
|
||||
"grade": _grade(score),
|
||||
"counts": counts,
|
||||
"categories": categories,
|
||||
}
|
||||
|
||||
|
||||
def _grade(score: int) -> str:
|
||||
if score >= 90:
|
||||
return "A"
|
||||
if score >= 80:
|
||||
return "B"
|
||||
if score >= 70:
|
||||
return "C"
|
||||
if score >= 55:
|
||||
return "D"
|
||||
return "F"
|
||||
@@ -0,0 +1,59 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .base import (
|
||||
MEDIUM,
|
||||
LOW,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
)
|
||||
|
||||
CATEGORY = "security"
|
||||
|
||||
SECURITY_HEADERS = {
|
||||
"content-security-policy": ("Content-Security-Policy", LOW),
|
||||
"x-content-type-options": ("X-Content-Type-Options", LOW),
|
||||
"x-frame-options": ("X-Frame-Options", LOW),
|
||||
"referrer-policy": ("Referrer-Policy", LOW),
|
||||
}
|
||||
|
||||
|
||||
@page_check
|
||||
def security_headers(page: PageContext, site: SiteContext) -> list:
|
||||
results = []
|
||||
for key, (label, severity) in SECURITY_HEADERS.items():
|
||||
present = bool(page.headers.get(key))
|
||||
results.append(
|
||||
ok_check(
|
||||
f"security.{key}",
|
||||
CATEGORY,
|
||||
label,
|
||||
present,
|
||||
severity=severity,
|
||||
value="present" if present else "missing",
|
||||
recommendation=f"Add the {label} response header.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
@page_check
|
||||
def tls_valid(page: PageContext, site: SiteContext):
|
||||
if "tlsValid" not in page.metrics:
|
||||
return None
|
||||
valid = bool(page.metrics.get("tlsValid"))
|
||||
return ok_check(
|
||||
"security.tls",
|
||||
CATEGORY,
|
||||
"Valid TLS certificate",
|
||||
valid,
|
||||
severity=MEDIUM,
|
||||
value="valid" if valid else "invalid",
|
||||
recommendation="Serve a valid, unexpired TLS certificate.",
|
||||
url=page.url,
|
||||
)
|
||||
@@ -0,0 +1,59 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .base import (
|
||||
MEDIUM,
|
||||
LOW,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
)
|
||||
|
||||
CATEGORY = "social"
|
||||
|
||||
OG_REQUIRED = ("og:title", "og:description", "og:image", "og:url")
|
||||
TWITTER_REQUIRED = ("twitter:card",)
|
||||
|
||||
|
||||
@page_check
|
||||
def open_graph(page: PageContext, site: SiteContext) -> list:
|
||||
og = page.dom.get("og", {}) or {}
|
||||
missing = [tag for tag in OG_REQUIRED if not og.get(tag)]
|
||||
results = [
|
||||
ok_check(
|
||||
"social.open_graph",
|
||||
CATEGORY,
|
||||
"Open Graph tags",
|
||||
not missing,
|
||||
severity=MEDIUM,
|
||||
value="complete" if not missing else f"missing {', '.join(missing)}",
|
||||
recommendation="Add og:title, og:description, og:image and og:url for rich social previews.",
|
||||
warn=True,
|
||||
details={"present": sorted(og.keys())},
|
||||
url=page.url,
|
||||
)
|
||||
]
|
||||
return results
|
||||
|
||||
|
||||
@page_check
|
||||
def twitter_card(page: PageContext, site: SiteContext) -> list:
|
||||
twitter = page.dom.get("twitter", {}) or {}
|
||||
og = page.dom.get("og", {}) or {}
|
||||
has_card = bool(twitter.get("twitter:card"))
|
||||
return [
|
||||
ok_check(
|
||||
"social.twitter_card",
|
||||
CATEGORY,
|
||||
"Twitter Card",
|
||||
has_card or bool(og),
|
||||
severity=LOW,
|
||||
value=twitter.get("twitter:card", "") or ("falls back to OG" if og else "missing"),
|
||||
recommendation="Add a twitter:card meta tag (summary_large_image) for X/Twitter previews.",
|
||||
warn=True,
|
||||
details={"present": sorted(twitter.keys())},
|
||||
url=page.url,
|
||||
)
|
||||
]
|
||||
@@ -0,0 +1,165 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from .base import (
|
||||
MEDIUM,
|
||||
LOW,
|
||||
INFO,
|
||||
PASS,
|
||||
WARN,
|
||||
FAIL,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
page_check,
|
||||
)
|
||||
|
||||
CATEGORY = "structured_data"
|
||||
|
||||
REQUIRED_PROPS = {
|
||||
"Article": ("headline",),
|
||||
"BlogPosting": ("headline",),
|
||||
"NewsArticle": ("headline",),
|
||||
"Product": ("name",),
|
||||
"Offer": ("price", "priceCurrency"),
|
||||
"Organization": ("name", "url"),
|
||||
"LocalBusiness": ("name", "address"),
|
||||
"BreadcrumbList": ("itemListElement",),
|
||||
"FAQPage": ("mainEntity",),
|
||||
"WebSite": ("name", "url"),
|
||||
"Person": ("name",),
|
||||
"Recipe": ("name", "recipeIngredient"),
|
||||
"Event": ("name", "startDate"),
|
||||
"VideoObject": ("name", "thumbnailUrl", "uploadDate"),
|
||||
"SoftwareApplication": ("name",),
|
||||
}
|
||||
|
||||
|
||||
def _iter_objects(parsed):
|
||||
if isinstance(parsed, list):
|
||||
for item in parsed:
|
||||
yield from _iter_objects(item)
|
||||
return
|
||||
if not isinstance(parsed, dict):
|
||||
return
|
||||
graph = parsed.get("@graph")
|
||||
if isinstance(graph, list):
|
||||
for item in graph:
|
||||
yield from _iter_objects(item)
|
||||
if parsed.get("@type"):
|
||||
yield parsed
|
||||
|
||||
|
||||
def _types(node) -> list:
|
||||
raw = node.get("@type")
|
||||
if isinstance(raw, list):
|
||||
return [str(item) for item in raw]
|
||||
return [str(raw)] if raw else []
|
||||
|
||||
|
||||
@page_check
|
||||
def jsonld(page: PageContext, site: SiteContext) -> list:
|
||||
blocks = page.dom.get("jsonld", []) or []
|
||||
present = bool(blocks)
|
||||
results = [
|
||||
ok_check(
|
||||
"structured.jsonld_present",
|
||||
CATEGORY,
|
||||
"Structured data (JSON-LD)",
|
||||
present,
|
||||
severity=MEDIUM,
|
||||
value=f"{len(blocks)} block(s)" if present else "none",
|
||||
recommendation="Add JSON-LD structured data so search engines can render rich results.",
|
||||
warn=True,
|
||||
url=page.url,
|
||||
)
|
||||
]
|
||||
if not present:
|
||||
return results
|
||||
|
||||
invalid = 0
|
||||
nodes = []
|
||||
for block in blocks:
|
||||
try:
|
||||
parsed = json.loads(block)
|
||||
except (ValueError, TypeError):
|
||||
invalid += 1
|
||||
continue
|
||||
nodes.extend(_iter_objects(parsed))
|
||||
|
||||
results.append(
|
||||
ok_check(
|
||||
"structured.jsonld_valid",
|
||||
CATEGORY,
|
||||
"JSON-LD parses",
|
||||
invalid == 0,
|
||||
severity=MEDIUM,
|
||||
value="valid" if invalid == 0 else f"{invalid} invalid block(s)",
|
||||
recommendation="Fix JSON syntax errors in structured-data blocks.",
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
|
||||
found_types = sorted({t for node in nodes for t in _types(node)})
|
||||
results.append(
|
||||
Check(
|
||||
"structured.types",
|
||||
CATEGORY,
|
||||
"Schema types",
|
||||
INFO,
|
||||
INFO,
|
||||
value=", ".join(found_types) or "none",
|
||||
details={"types": found_types},
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
|
||||
missing = []
|
||||
for node in nodes:
|
||||
for type_name in _types(node):
|
||||
required = REQUIRED_PROPS.get(type_name)
|
||||
if not required:
|
||||
continue
|
||||
absent = [prop for prop in required if not node.get(prop)]
|
||||
if absent:
|
||||
missing.append(f"{type_name}: {', '.join(absent)}")
|
||||
if missing:
|
||||
results.append(
|
||||
Check(
|
||||
"structured.required_props",
|
||||
CATEGORY,
|
||||
"Required schema properties",
|
||||
WARN,
|
||||
LOW,
|
||||
value=f"{len(missing)} type(s) missing props",
|
||||
recommendation="Add the required properties for each declared schema type.",
|
||||
details={"missing": missing[:20]},
|
||||
url=page.url,
|
||||
)
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
@page_check
|
||||
def other_markup(page: PageContext, site: SiteContext) -> Check:
|
||||
microdata = bool(page.dom.get("microdata"))
|
||||
rdfa = bool(page.dom.get("rdfa"))
|
||||
formats = []
|
||||
if microdata:
|
||||
formats.append("microdata")
|
||||
if rdfa:
|
||||
formats.append("RDFa")
|
||||
return Check(
|
||||
"structured.other_markup",
|
||||
CATEGORY,
|
||||
"Other structured-data formats",
|
||||
INFO,
|
||||
INFO,
|
||||
value=", ".join(formats) or "none",
|
||||
details={"microdata": microdata, "rdfa": rdfa},
|
||||
url=page.url,
|
||||
)
|
||||
@@ -0,0 +1,422 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from urllib.parse import urljoin, urlparse
|
||||
from xml.etree import ElementTree
|
||||
|
||||
import httpx
|
||||
|
||||
from devplacepy.net_guard import BlockedAddressError, guard_public_url
|
||||
from .checks.base import PageContext, SiteContext
|
||||
|
||||
USER_AGENT = (
|
||||
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/131.0.0.0 Safari/537.36 DevPlaceSEOBot/1.0"
|
||||
)
|
||||
NAV_TIMEOUT_MS = 30000
|
||||
RAW_FETCH_TIMEOUT = 15.0
|
||||
MAX_RAW_BYTES = 3_000_000
|
||||
MOBILE_VIEWPORT = {"width": 390, "height": 844}
|
||||
DESKTOP_VIEWPORT = {"width": 1366, "height": 900}
|
||||
|
||||
INIT_SCRIPT = """
|
||||
window.__seo = { lcp: 0, cls: 0, consoleErrors: 0, errorSamples: [] };
|
||||
try {
|
||||
new PerformanceObserver((list) => {
|
||||
for (const entry of list.getEntries()) {
|
||||
window.__seo.lcp = Math.max(window.__seo.lcp, entry.startTime || entry.renderTime || 0);
|
||||
}
|
||||
}).observe({ type: 'largest-contentful-paint', buffered: true });
|
||||
} catch (e) {}
|
||||
try {
|
||||
new PerformanceObserver((list) => {
|
||||
for (const entry of list.getEntries()) {
|
||||
if (!entry.hadRecentInput) window.__seo.cls += entry.value;
|
||||
}
|
||||
}).observe({ type: 'layout-shift', buffered: true });
|
||||
} catch (e) {}
|
||||
"""
|
||||
|
||||
EXTRACT_SCRIPT = r"""
|
||||
() => {
|
||||
const abs = (href) => { try { return new URL(href, document.baseURI).href; } catch (e) { return href || ''; } };
|
||||
const metaByName = (name) => {
|
||||
const el = document.querySelector(`meta[name="${name}"]`);
|
||||
return el ? (el.getAttribute('content') || '') : '';
|
||||
};
|
||||
const og = {};
|
||||
document.querySelectorAll('meta[property^="og:"]').forEach((m) => {
|
||||
og[m.getAttribute('property')] = m.getAttribute('content') || '';
|
||||
});
|
||||
const twitter = {};
|
||||
document.querySelectorAll('meta[name^="twitter:"]').forEach((m) => {
|
||||
twitter[m.getAttribute('name')] = m.getAttribute('content') || '';
|
||||
});
|
||||
const headings = [];
|
||||
document.querySelectorAll('h1,h2,h3,h4,h5,h6').forEach((h) => {
|
||||
headings.push({ level: parseInt(h.tagName.substring(1), 10), text: (h.textContent || '').trim().slice(0, 160) });
|
||||
});
|
||||
const h1 = Array.from(document.querySelectorAll('h1')).map((h) => (h.textContent || '').trim().slice(0, 160));
|
||||
const images = Array.from(document.querySelectorAll('img')).slice(0, 200).map((img) => ({
|
||||
src: abs(img.getAttribute('src') || ''),
|
||||
alt: img.getAttribute('alt'),
|
||||
width: img.getAttribute('width') || (img.naturalWidth ? String(img.naturalWidth) : ''),
|
||||
height: img.getAttribute('height') || (img.naturalHeight ? String(img.naturalHeight) : ''),
|
||||
loading: img.getAttribute('loading') || ''
|
||||
}));
|
||||
const links = Array.from(document.querySelectorAll('a[href]')).slice(0, 500).map((a) => ({
|
||||
href: abs(a.getAttribute('href') || ''),
|
||||
rel: a.getAttribute('rel') || '',
|
||||
text: (a.textContent || '').trim().slice(0, 120)
|
||||
}));
|
||||
const jsonld = Array.from(document.querySelectorAll('script[type="application/ld+json"]')).map((s) => s.textContent || '');
|
||||
const hreflang = Array.from(document.querySelectorAll('link[rel="alternate"][hreflang]')).map((l) => ({
|
||||
hreflang: l.getAttribute('hreflang'), href: abs(l.getAttribute('href') || '')
|
||||
}));
|
||||
const canonicalEls = document.querySelectorAll('link[rel="canonical"]');
|
||||
const isHttps = location.protocol === 'https:';
|
||||
const mixed = [];
|
||||
if (isHttps) {
|
||||
document.querySelectorAll('[src],[href]').forEach((el) => {
|
||||
const v = el.getAttribute('src') || el.getAttribute('href') || '';
|
||||
if (v.startsWith('http://')) mixed.push(v);
|
||||
});
|
||||
}
|
||||
const viewportEl = document.querySelector('meta[name="viewport"]');
|
||||
const charsetEl = document.querySelector('meta[charset]');
|
||||
let formFieldCount = 0;
|
||||
let formsMissingLabels = 0;
|
||||
document.querySelectorAll('input,select,textarea').forEach((field) => {
|
||||
const type = (field.getAttribute('type') || '').toLowerCase();
|
||||
if (type === 'hidden' || type === 'submit' || type === 'button') return;
|
||||
formFieldCount += 1;
|
||||
const id = field.getAttribute('id');
|
||||
const labelled = (id && document.querySelector(`label[for="${id}"]`)) ||
|
||||
field.getAttribute('aria-label') || field.getAttribute('aria-labelledby') ||
|
||||
field.closest('label');
|
||||
if (!labelled) formsMissingLabels += 1;
|
||||
});
|
||||
const nav = performance.getEntriesByType('navigation')[0] || {};
|
||||
const resources = performance.getEntriesByType('resource') || [];
|
||||
let transfer = nav.transferSize || 0;
|
||||
resources.forEach((r) => { transfer += (r.transferSize || 0); });
|
||||
const bodyText = (document.body ? document.body.innerText || '' : '');
|
||||
const wordCount = bodyText.split(/\s+/).filter(Boolean).length;
|
||||
return {
|
||||
title: (document.title || '').trim(),
|
||||
titleCount: document.querySelectorAll('title').length,
|
||||
metaDescription: metaByName('description'),
|
||||
metaRobots: metaByName('robots'),
|
||||
canonical: canonicalEls.length ? abs(canonicalEls[0].getAttribute('href') || '') : '',
|
||||
canonicalCount: canonicalEls.length,
|
||||
htmlLang: document.documentElement.getAttribute('lang') || '',
|
||||
charset: charsetEl ? (charsetEl.getAttribute('charset') || '') : (document.characterSet || ''),
|
||||
hasViewport: !!viewportEl,
|
||||
viewportContent: viewportEl ? (viewportEl.getAttribute('content') || '') : '',
|
||||
favicon: !!document.querySelector('link[rel~="icon"]'),
|
||||
h1: h1,
|
||||
headings: headings,
|
||||
images: images,
|
||||
links: links,
|
||||
jsonld: jsonld,
|
||||
hreflang: hreflang,
|
||||
microdata: !!document.querySelector('[itemscope]'),
|
||||
rdfa: !!document.querySelector('[vocab],[typeof],[property]'),
|
||||
og: og,
|
||||
twitter: twitter,
|
||||
iframeCount: document.querySelectorAll('iframe').length,
|
||||
semantic: {
|
||||
main: document.querySelectorAll('main').length,
|
||||
article: document.querySelectorAll('article').length,
|
||||
nav: document.querySelectorAll('nav').length,
|
||||
header: document.querySelectorAll('header').length,
|
||||
footer: document.querySelectorAll('footer').length
|
||||
},
|
||||
mixedContent: mixed,
|
||||
formFieldCount: formFieldCount,
|
||||
formsMissingLabels: formsMissingLabels,
|
||||
wordCount: wordCount,
|
||||
metrics: {
|
||||
ttfb: Math.max(0, (nav.responseStart || 0) - (nav.requestStart || 0)),
|
||||
fcp: (performance.getEntriesByName('first-contentful-paint')[0] || {}).startTime || 0,
|
||||
domContentLoaded: nav.domContentLoadedEventEnd || 0,
|
||||
load: nav.loadEventEnd || 0,
|
||||
transferSize: transfer,
|
||||
resourceCount: resources.length,
|
||||
domNodes: document.getElementsByTagName('*').length,
|
||||
protocol: nav.nextHopProtocol || '',
|
||||
lcp: window.__seo ? window.__seo.lcp : 0,
|
||||
cls: window.__seo ? window.__seo.cls : 0
|
||||
}
|
||||
};
|
||||
}
|
||||
"""
|
||||
|
||||
MOBILE_SCRIPT = r"""
|
||||
() => {
|
||||
let small = 0;
|
||||
document.querySelectorAll('a,button,[role="button"],input[type="submit"]').forEach((el) => {
|
||||
const r = el.getBoundingClientRect();
|
||||
if (r.width > 0 && r.height > 0 && (r.width < 24 || r.height < 24)) small += 1;
|
||||
});
|
||||
return {
|
||||
scrollWidth: document.documentElement.scrollWidth,
|
||||
clientWidth: document.documentElement.clientWidth,
|
||||
hasHorizontalOverflow: document.documentElement.scrollWidth > document.documentElement.clientWidth + 4,
|
||||
smallTapTargets: small
|
||||
};
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
def _normalize_url(url: str) -> str:
|
||||
url = (url or "").strip()
|
||||
if not url:
|
||||
return ""
|
||||
if "://" not in url:
|
||||
url = "https://" + url
|
||||
return url
|
||||
|
||||
|
||||
async def _fetch_text(client: httpx.AsyncClient, url: str) -> dict:
|
||||
try:
|
||||
response = await client.get(url)
|
||||
body = response.text[:MAX_RAW_BYTES]
|
||||
return {"status": response.status_code, "text": body, "url": str(response.url)}
|
||||
except httpx.HTTPError as exc:
|
||||
return {"status": 0, "text": "", "url": url, "error": str(exc)[:200]}
|
||||
|
||||
|
||||
def _parse_robots(text: str, target_path: str) -> dict:
|
||||
disallows = []
|
||||
sitemap_urls = []
|
||||
active = False
|
||||
for raw in text.splitlines():
|
||||
line = raw.split("#", 1)[0].strip()
|
||||
if not line:
|
||||
continue
|
||||
key, _, value = line.partition(":")
|
||||
key = key.strip().lower()
|
||||
value = value.strip()
|
||||
if key == "user-agent":
|
||||
active = value == "*"
|
||||
elif key == "sitemap":
|
||||
sitemap_urls.append(value)
|
||||
elif key == "disallow" and active and value:
|
||||
disallows.append(value)
|
||||
blocks_target = any(
|
||||
target_path.startswith(rule) for rule in disallows if rule and rule != "/"
|
||||
) or "/" in [r for r in disallows]
|
||||
return {
|
||||
"disallows": disallows,
|
||||
"sitemap_urls": sitemap_urls,
|
||||
"blocks_target": blocks_target,
|
||||
}
|
||||
|
||||
|
||||
def _parse_sitemap(text: str) -> dict:
|
||||
result = {"valid_xml": False, "url_count": 0, "lastmod_count": 0, "urls": []}
|
||||
try:
|
||||
root = ElementTree.fromstring(text)
|
||||
except ElementTree.ParseError:
|
||||
return result
|
||||
result["valid_xml"] = True
|
||||
urls = []
|
||||
lastmods = 0
|
||||
for url_el in root.iter():
|
||||
tag = url_el.tag.rsplit("}", 1)[-1]
|
||||
if tag == "loc" and url_el.text:
|
||||
urls.append(url_el.text.strip())
|
||||
elif tag == "lastmod" and url_el.text:
|
||||
lastmods += 1
|
||||
result["url_count"] = len(urls)
|
||||
result["lastmod_count"] = lastmods
|
||||
result["urls"] = urls
|
||||
return result
|
||||
|
||||
|
||||
async def _site_resources(
|
||||
client: httpx.AsyncClient, base_url: str, sitemap_url: str, target_path: str
|
||||
) -> tuple[dict, dict, dict]:
|
||||
parsed = urlparse(base_url)
|
||||
root = f"{parsed.scheme}://{parsed.netloc}"
|
||||
robots_raw = await _fetch_text(client, urljoin(root + "/", "robots.txt"))
|
||||
robots = dict(robots_raw)
|
||||
if robots_raw.get("status") == 200:
|
||||
robots.update(_parse_robots(robots_raw.get("text", ""), target_path))
|
||||
sm_target = sitemap_url or urljoin(root + "/", "sitemap.xml")
|
||||
sitemap_raw = await _fetch_text(client, sm_target)
|
||||
sitemap = dict(sitemap_raw)
|
||||
if sitemap_raw.get("status") == 200:
|
||||
sitemap.update(_parse_sitemap(sitemap_raw.get("text", "")))
|
||||
llms_raw = await _fetch_text(client, urljoin(root + "/", "llms.txt"))
|
||||
llms = {"found": llms_raw.get("status") == 200, "status": llms_raw.get("status")}
|
||||
return robots, sitemap, llms
|
||||
|
||||
|
||||
async def _build_page(context, client, url: str, output_dir, index: int) -> PageContext:
|
||||
page_ctx = PageContext(requested_url=url, url=url)
|
||||
page = await context.new_page()
|
||||
console_errors = []
|
||||
|
||||
def _on_console(message):
|
||||
if message.type == "error":
|
||||
console_errors.append(message.text[:200])
|
||||
|
||||
page.on("console", _on_console)
|
||||
page.on("pageerror", lambda exc: console_errors.append(str(exc)[:200]))
|
||||
await page.add_init_script(INIT_SCRIPT)
|
||||
try:
|
||||
response = await page.goto(url, wait_until="load", timeout=NAV_TIMEOUT_MS)
|
||||
try:
|
||||
await page.wait_for_load_state("networkidle", timeout=8000)
|
||||
except Exception:
|
||||
pass
|
||||
page_ctx.status = response.status if response else 0
|
||||
page_ctx.url = page.url
|
||||
page_ctx.ok = bool(response and response.ok)
|
||||
if response is not None:
|
||||
page_ctx.headers = {k.lower(): v for k, v in response.headers.items()}
|
||||
chain = []
|
||||
req = response.request.redirected_from
|
||||
while req is not None:
|
||||
resp = await req.response()
|
||||
chain.append({"url": req.url, "status": resp.status if resp else 0})
|
||||
req = req.redirected_from
|
||||
page_ctx.redirect_chain = list(reversed(chain))
|
||||
dom = await page.evaluate(EXTRACT_SCRIPT)
|
||||
page_ctx.metrics = dom.pop("metrics", {})
|
||||
page_ctx.metrics["consoleErrors"] = len(console_errors)
|
||||
page_ctx.metrics["consoleErrorSamples"] = console_errors[:5]
|
||||
page_ctx.dom = dom
|
||||
page_ctx.rendered_html = (await page.content())[:MAX_RAW_BYTES]
|
||||
try:
|
||||
await page.set_viewport_size(MOBILE_VIEWPORT)
|
||||
page_ctx.mobile = await page.evaluate(MOBILE_SCRIPT)
|
||||
await page.set_viewport_size(DESKTOP_VIEWPORT)
|
||||
except Exception:
|
||||
page_ctx.mobile = {}
|
||||
if output_dir is not None:
|
||||
shot = output_dir / f"page-{index}.png"
|
||||
try:
|
||||
await page.screenshot(path=str(shot), full_page=False)
|
||||
page_ctx.screenshot = shot.name
|
||||
except Exception:
|
||||
page_ctx.screenshot = ""
|
||||
raw = await _fetch_text(client, page_ctx.url)
|
||||
page_ctx.raw_html = raw.get("text", "")
|
||||
except Exception as exc: # noqa: BLE001 - record the failure as page state
|
||||
page_ctx.error = str(exc)[:300]
|
||||
page_ctx.ok = False
|
||||
finally:
|
||||
await page.close()
|
||||
return page_ctx
|
||||
|
||||
|
||||
async def crawl_target(payload: dict, emit, output_dir) -> SiteContext:
|
||||
target = _normalize_url(payload.get("url", ""))
|
||||
mode = payload.get("mode", "url")
|
||||
allow_private = bool(payload.get("allow_private"))
|
||||
max_pages = max(1, min(int(payload.get("max_pages", 1) or 1), 50))
|
||||
|
||||
await guard_public_url(target, allow_private=allow_private)
|
||||
parsed = urlparse(target)
|
||||
site = SiteContext(
|
||||
target_url=target,
|
||||
mode=mode,
|
||||
base_host=parsed.netloc,
|
||||
base_scheme=parsed.scheme,
|
||||
)
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async with httpx.AsyncClient(
|
||||
follow_redirects=True,
|
||||
timeout=RAW_FETCH_TIMEOUT,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
) as client:
|
||||
emit({"type": "stage", "stage": "resources", "message": "Reading robots.txt and sitemap"})
|
||||
sitemap_hint = target if mode == "sitemap" else ""
|
||||
robots, sitemap, llms = await _site_resources(
|
||||
client, target, sitemap_hint, parsed.path or "/"
|
||||
)
|
||||
site.robots = robots
|
||||
site.sitemap = sitemap
|
||||
site.llms_txt = llms
|
||||
|
||||
if mode == "sitemap":
|
||||
candidates = sitemap.get("urls", [])[:max_pages]
|
||||
else:
|
||||
candidates = [target]
|
||||
|
||||
safe_urls = []
|
||||
for url in candidates:
|
||||
host = urlparse(url).netloc
|
||||
if host and host == site.base_host:
|
||||
# Same host as the audited target, already validated by the
|
||||
# top-level guard above; skip the redundant per-URL DNS lookup.
|
||||
safe_urls.append(url)
|
||||
continue
|
||||
try:
|
||||
await guard_public_url(url, allow_private=allow_private)
|
||||
safe_urls.append(url)
|
||||
except BlockedAddressError:
|
||||
continue
|
||||
|
||||
if not safe_urls:
|
||||
if mode == "sitemap":
|
||||
raise ValueError(
|
||||
"No crawlable page URLs found in the sitemap "
|
||||
f"({sitemap.get('url_count', 0)} entries; none reachable or all blocked)."
|
||||
)
|
||||
safe_urls = [target]
|
||||
|
||||
emit(
|
||||
{
|
||||
"type": "target",
|
||||
"url": target,
|
||||
"mode": mode,
|
||||
"urls": safe_urls,
|
||||
"sitemap_total": sitemap.get("url_count", 0) if mode == "sitemap" else 0,
|
||||
}
|
||||
)
|
||||
|
||||
async with async_playwright() as pw:
|
||||
browser = await pw.chromium.launch(
|
||||
headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"]
|
||||
)
|
||||
context = await browser.new_context(
|
||||
viewport=DESKTOP_VIEWPORT,
|
||||
user_agent=USER_AGENT,
|
||||
ignore_https_errors=False,
|
||||
)
|
||||
try:
|
||||
total = len(safe_urls)
|
||||
for index, url in enumerate(safe_urls):
|
||||
emit(
|
||||
{
|
||||
"type": "progress",
|
||||
"done": index,
|
||||
"total": total,
|
||||
"url": url,
|
||||
"message": f"Auditing {url}",
|
||||
}
|
||||
)
|
||||
page_ctx = await _build_page(
|
||||
context, client, url, output_dir, index
|
||||
)
|
||||
site.pages.append(page_ctx)
|
||||
yield_frame = {
|
||||
"type": "page_loaded",
|
||||
"url": page_ctx.url,
|
||||
"status": page_ctx.status,
|
||||
"done": index + 1,
|
||||
"total": total,
|
||||
}
|
||||
emit(yield_frame)
|
||||
finally:
|
||||
await context.close()
|
||||
await browser.close()
|
||||
return site
|
||||
@@ -0,0 +1,43 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
MAX_BUFFER = 2000
|
||||
|
||||
|
||||
class ProgressHub:
|
||||
def __init__(self) -> None:
|
||||
self._subscribers: dict[str, set[asyncio.Queue]] = {}
|
||||
self._buffers: dict[str, list[dict]] = {}
|
||||
|
||||
def subscribe(self, uid: str) -> asyncio.Queue:
|
||||
queue: asyncio.Queue = asyncio.Queue()
|
||||
self._subscribers.setdefault(uid, set()).add(queue)
|
||||
return queue
|
||||
|
||||
def unsubscribe(self, uid: str, queue: asyncio.Queue) -> None:
|
||||
listeners = self._subscribers.get(uid)
|
||||
if not listeners:
|
||||
return
|
||||
listeners.discard(queue)
|
||||
if not listeners:
|
||||
self._subscribers.pop(uid, None)
|
||||
|
||||
def publish(self, uid: str, frame: dict) -> None:
|
||||
buffer = self._buffers.setdefault(uid, [])
|
||||
buffer.append(frame)
|
||||
if len(buffer) > MAX_BUFFER:
|
||||
del buffer[: len(buffer) - MAX_BUFFER]
|
||||
for queue in self._subscribers.get(uid, set()):
|
||||
queue.put_nowait(frame)
|
||||
|
||||
def snapshot(self, uid: str) -> list[dict]:
|
||||
return list(self._buffers.get(uid, []))
|
||||
|
||||
def clear(self, uid: str) -> None:
|
||||
self._buffers.pop(uid, None)
|
||||
|
||||
|
||||
hub = ProgressHub()
|
||||
@@ -0,0 +1,155 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import shutil
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from devplacepy.config import BASE_DIR, SEO_REPORTS_DIR
|
||||
from devplacepy.services.jobs.base import JobService
|
||||
from .progress import hub
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
WORKER_MODULE = "devplacepy.services.jobs.seo.worker"
|
||||
STREAM_LIMIT = 16 * 1024 * 1024
|
||||
|
||||
|
||||
class SeoService(JobService):
|
||||
kind = "seo"
|
||||
title = "SEO Diagnostics"
|
||||
description = (
|
||||
"Audits any URL or sitemap with a broad battery of technical, on-page, structured-data, "
|
||||
"performance, accessibility and AI-readiness checks using a headless browser, streaming "
|
||||
"live progress over a websocket and storing a categorised report."
|
||||
)
|
||||
|
||||
def __init__(self):
|
||||
super().__init__(name="seo", interval_seconds=2)
|
||||
|
||||
def report_dir(self, uid: str) -> Path:
|
||||
return SEO_REPORTS_DIR / uid
|
||||
|
||||
async def process(self, job: dict) -> dict:
|
||||
from devplacepy.services.audit import record as audit
|
||||
|
||||
uid = job["uid"]
|
||||
payload = job.get("payload", {})
|
||||
target = payload.get("url", "")
|
||||
output_dir = self.report_dir(uid)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
payload_path = output_dir / "payload.json"
|
||||
payload_path.write_text(json.dumps(payload), encoding="utf-8")
|
||||
|
||||
actor_kind = "user" if job.get("owner_kind") == "user" else (
|
||||
job.get("owner_kind") or "system"
|
||||
)
|
||||
actor_uid = job.get("owner_id") if job.get("owner_kind") == "user" else None
|
||||
|
||||
try:
|
||||
summary = await self._run_worker(uid, payload_path, output_dir)
|
||||
except Exception as exc:
|
||||
hub.publish(uid, {"type": "failed", "message": str(exc)[:300]})
|
||||
audit.record_system(
|
||||
"seo.run.failed",
|
||||
actor_kind=actor_kind,
|
||||
actor_uid=actor_uid,
|
||||
result="failure",
|
||||
summary=f"SEO diagnostics for {target} failed",
|
||||
metadata={"target": target, "error": str(exc)[:200]},
|
||||
links=[audit.job(uid)],
|
||||
)
|
||||
raise
|
||||
|
||||
report = self._load_report(output_dir)
|
||||
hub.publish(
|
||||
uid,
|
||||
{
|
||||
"type": "done",
|
||||
"score": summary.get("score"),
|
||||
"grade": summary.get("grade"),
|
||||
"page_count": summary.get("page_count"),
|
||||
"report_url": f"/tools/seo/{uid}/report",
|
||||
"report": report,
|
||||
},
|
||||
)
|
||||
hub.clear(uid)
|
||||
audit.record_system(
|
||||
"seo.run.complete",
|
||||
actor_kind=actor_kind,
|
||||
actor_uid=actor_uid,
|
||||
summary=f"SEO diagnostics for {target} scored {summary.get('score')}",
|
||||
metadata={
|
||||
"target": target,
|
||||
"score": summary.get("score"),
|
||||
"grade": summary.get("grade"),
|
||||
"page_count": summary.get("page_count"),
|
||||
},
|
||||
links=[audit.job(uid)],
|
||||
)
|
||||
return {
|
||||
"target": target,
|
||||
"mode": payload.get("mode", "url"),
|
||||
"score": summary.get("score", 0),
|
||||
"grade": summary.get("grade", ""),
|
||||
"page_count": summary.get("page_count", 0),
|
||||
"counts": summary.get("counts", {}),
|
||||
"categories": summary.get("categories", {}),
|
||||
"report": report,
|
||||
"report_url": f"/tools/seo/{uid}/report",
|
||||
"bytes_in": 0,
|
||||
"bytes_out": len(json.dumps(report)) if report else 0,
|
||||
"item_count": summary.get("page_count", 0),
|
||||
}
|
||||
|
||||
async def _run_worker(self, uid: str, payload_path: Path, output_dir: Path) -> dict:
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
sys.executable,
|
||||
"-m",
|
||||
WORKER_MODULE,
|
||||
str(payload_path),
|
||||
str(output_dir),
|
||||
cwd=str(BASE_DIR),
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
limit=STREAM_LIMIT,
|
||||
)
|
||||
summary: dict = {}
|
||||
worker_error = ""
|
||||
while True:
|
||||
line = await proc.stdout.readline()
|
||||
if not line:
|
||||
break
|
||||
try:
|
||||
frame = json.loads(line.decode("utf-8", "replace"))
|
||||
except (ValueError, TypeError):
|
||||
continue
|
||||
hub.publish(uid, frame)
|
||||
if frame.get("type") == "report_ready":
|
||||
summary = frame
|
||||
elif frame.get("type") == "error":
|
||||
worker_error = frame.get("message", "worker error")
|
||||
err = (await proc.stderr.read()).decode("utf-8", "replace")
|
||||
await proc.wait()
|
||||
if proc.returncode != 0 or not summary:
|
||||
raise RuntimeError(
|
||||
worker_error or err[:500] or f"seo worker exited {proc.returncode}"
|
||||
)
|
||||
return summary
|
||||
|
||||
def _load_report(self, output_dir: Path) -> dict:
|
||||
report_path = output_dir / "report.json"
|
||||
if not report_path.is_file():
|
||||
return {}
|
||||
try:
|
||||
return json.loads(report_path.read_text(encoding="utf-8"))
|
||||
except (ValueError, OSError):
|
||||
return {}
|
||||
|
||||
def cleanup(self, job: dict) -> None:
|
||||
hub.clear(job["uid"])
|
||||
shutil.rmtree(self.report_dir(job["uid"]), ignore_errors=True)
|
||||
@@ -0,0 +1,132 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from .checks.registry import compute_score, run_page_checks, run_site_checks
|
||||
from .crawler import crawl_target
|
||||
|
||||
|
||||
def _emit(frame: dict) -> None:
|
||||
sys.stdout.write(json.dumps(frame, ensure_ascii=False) + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def _trim_site(site) -> dict:
|
||||
robots = site.robots or {}
|
||||
sitemap = site.sitemap or {}
|
||||
return {
|
||||
"robots": {
|
||||
"status": robots.get("status"),
|
||||
"disallows": robots.get("disallows", [])[:50],
|
||||
"sitemap_urls": robots.get("sitemap_urls", [])[:10],
|
||||
"blocks_target": robots.get("blocks_target", False),
|
||||
},
|
||||
"sitemap": {
|
||||
"status": sitemap.get("status"),
|
||||
"valid_xml": sitemap.get("valid_xml", False),
|
||||
"url_count": sitemap.get("url_count", 0),
|
||||
"lastmod_count": sitemap.get("lastmod_count", 0),
|
||||
},
|
||||
"llms_txt": site.llms_txt or {},
|
||||
}
|
||||
|
||||
|
||||
async def _run(payload: dict, output_dir: Path) -> dict:
|
||||
site = await crawl_target(payload, _emit, output_dir)
|
||||
|
||||
all_checks = []
|
||||
pages_summary = []
|
||||
_emit({"type": "stage", "stage": "checks", "message": "Running SEO checks"})
|
||||
for page in site.pages:
|
||||
page_checks = run_page_checks(page, site)
|
||||
all_checks.extend(page_checks)
|
||||
page_score = compute_score(page_checks)
|
||||
page_dicts = [c.to_dict() for c in page_checks]
|
||||
pages_summary.append(
|
||||
{
|
||||
"url": page.url,
|
||||
"requested_url": page.requested_url,
|
||||
"status": page.status,
|
||||
"ok": page.ok,
|
||||
"error": page.error,
|
||||
"screenshot": page.screenshot,
|
||||
"score": page_score["score"],
|
||||
"grade": page_score["grade"],
|
||||
"counts": page_score["counts"],
|
||||
}
|
||||
)
|
||||
_emit(
|
||||
{
|
||||
"type": "page",
|
||||
"url": page.url,
|
||||
"status": page.status,
|
||||
"score": page_score["score"],
|
||||
"grade": page_score["grade"],
|
||||
"counts": page_score["counts"],
|
||||
"checks": page_dicts,
|
||||
}
|
||||
)
|
||||
|
||||
site_checks = run_site_checks(site)
|
||||
all_checks.extend(site_checks)
|
||||
_emit(
|
||||
{
|
||||
"type": "site_checks",
|
||||
"checks": [c.to_dict() for c in site_checks],
|
||||
}
|
||||
)
|
||||
|
||||
overall = compute_score(all_checks)
|
||||
report = {
|
||||
"target": site.target_url,
|
||||
"mode": site.mode,
|
||||
"generated_at": datetime.now(timezone.utc).isoformat(),
|
||||
"page_count": len(site.pages),
|
||||
"score": overall["score"],
|
||||
"grade": overall["grade"],
|
||||
"counts": overall["counts"],
|
||||
"categories": overall["categories"],
|
||||
"pages": pages_summary,
|
||||
"checks": [c.to_dict() for c in all_checks],
|
||||
"site": _trim_site(site),
|
||||
}
|
||||
if output_dir is not None:
|
||||
(output_dir / "report.json").write_text(
|
||||
json.dumps(report, ensure_ascii=False), encoding="utf-8"
|
||||
)
|
||||
_emit(
|
||||
{
|
||||
"type": "report_ready",
|
||||
"score": report["score"],
|
||||
"grade": report["grade"],
|
||||
"counts": report["counts"],
|
||||
"categories": report["categories"],
|
||||
"page_count": report["page_count"],
|
||||
}
|
||||
)
|
||||
return report
|
||||
|
||||
|
||||
def main(argv: list) -> int:
|
||||
if len(argv) != 3:
|
||||
sys.stderr.write("usage: seo.worker <payload_json> <output_dir>\n")
|
||||
return 2
|
||||
payload = json.loads(Path(argv[1]).read_text(encoding="utf-8"))
|
||||
output_dir = Path(argv[2])
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
try:
|
||||
asyncio.run(_run(payload, output_dir))
|
||||
except Exception as exc: # noqa: BLE001 - surface as a worker error line
|
||||
_emit({"type": "error", "message": str(exc)[:500]})
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main(sys.argv))
|
||||
Reference in New Issue
Block a user