forked from retoor/devplacepy
chore: replace python -m agents.validator with hawk in all agent mode instructions
This commit is contained in:
@@ -6,6 +6,7 @@ from .base import (
|
||||
HIGH,
|
||||
MEDIUM,
|
||||
LOW,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
@@ -16,7 +17,7 @@ CATEGORY = "mobile_accessibility"
|
||||
|
||||
|
||||
@page_check
|
||||
def responsive(page: PageContext, site: SiteContext) -> PageContext:
|
||||
def responsive(page: PageContext, site: SiteContext) -> Check:
|
||||
mobile = page.mobile or {}
|
||||
overflow = bool(mobile.get("hasHorizontalOverflow"))
|
||||
return ok_check(
|
||||
@@ -33,7 +34,7 @@ def responsive(page: PageContext, site: SiteContext) -> PageContext:
|
||||
|
||||
|
||||
@page_check
|
||||
def tap_targets(page: PageContext, site: SiteContext):
|
||||
def tap_targets(page: PageContext, site: SiteContext) -> Check | None:
|
||||
mobile = page.mobile or {}
|
||||
small = int(mobile.get("smallTapTargets", 0) or 0)
|
||||
if "smallTapTargets" not in mobile:
|
||||
@@ -52,7 +53,7 @@ def tap_targets(page: PageContext, site: SiteContext):
|
||||
|
||||
|
||||
@page_check
|
||||
def image_alt(page: PageContext, site: SiteContext):
|
||||
def image_alt(page: PageContext, site: SiteContext) -> Check | None:
|
||||
images = page.dom.get("images", []) or []
|
||||
if not images:
|
||||
return None
|
||||
@@ -71,7 +72,7 @@ def image_alt(page: PageContext, site: SiteContext):
|
||||
|
||||
|
||||
@page_check
|
||||
def form_labels(page: PageContext, site: SiteContext):
|
||||
def form_labels(page: PageContext, site: SiteContext) -> Check | None:
|
||||
missing = int(page.dom.get("formsMissingLabels", 0) or 0)
|
||||
if not page.dom.get("formFieldCount"):
|
||||
return None
|
||||
@@ -89,7 +90,7 @@ def form_labels(page: PageContext, site: SiteContext):
|
||||
|
||||
|
||||
@page_check
|
||||
def document_title(page: PageContext, site: SiteContext):
|
||||
def document_title(page: PageContext, site: SiteContext) -> Check:
|
||||
return ok_check(
|
||||
"a11y.document_title",
|
||||
CATEGORY,
|
||||
|
||||
@@ -26,7 +26,7 @@ from . import ai_readiness # noqa: F401
|
||||
from . import crosspage # noqa: F401
|
||||
|
||||
|
||||
def _collect(result) -> list[Check]:
|
||||
def _collect(result: object) -> list[Check]:
|
||||
if result is None:
|
||||
return []
|
||||
if isinstance(result, Check):
|
||||
|
||||
@@ -5,6 +5,7 @@ from __future__ import annotations
|
||||
from .base import (
|
||||
MEDIUM,
|
||||
LOW,
|
||||
Check,
|
||||
PageContext,
|
||||
SiteContext,
|
||||
ok_check,
|
||||
@@ -43,7 +44,7 @@ def security_headers(page: PageContext, site: SiteContext) -> list:
|
||||
|
||||
|
||||
@page_check
|
||||
def tls_valid(page: PageContext, site: SiteContext):
|
||||
def tls_valid(page: PageContext, site: SiteContext) -> Check | None:
|
||||
if "tlsValid" not in page.metrics:
|
||||
return None
|
||||
valid = bool(page.metrics.get("tlsValid"))
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Iterator
|
||||
|
||||
from .base import (
|
||||
MEDIUM,
|
||||
@@ -39,7 +40,7 @@ REQUIRED_PROPS = {
|
||||
}
|
||||
|
||||
|
||||
def _iter_objects(parsed):
|
||||
def _iter_objects(parsed: object) -> Iterator[dict]:
|
||||
if isinstance(parsed, list):
|
||||
for item in parsed:
|
||||
yield from _iter_objects(item)
|
||||
@@ -54,7 +55,7 @@ def _iter_objects(parsed):
|
||||
yield parsed
|
||||
|
||||
|
||||
def _types(node) -> list:
|
||||
def _types(node: dict) -> list:
|
||||
raw = node.get("@type")
|
||||
if isinstance(raw, list):
|
||||
return [str(item) for item in raw]
|
||||
|
||||
@@ -3,12 +3,18 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
from urllib.parse import urljoin, urlparse
|
||||
from xml.etree import ElementTree
|
||||
|
||||
import httpx
|
||||
|
||||
from devplacepy.net_guard import BlockedAddressError, guard_public_url
|
||||
from devplacepy.net_guard import (
|
||||
BlockedAddressError,
|
||||
guard_public_url,
|
||||
guarded_async_client,
|
||||
)
|
||||
from .checks.base import PageContext, SiteContext
|
||||
|
||||
USER_AGENT = (
|
||||
@@ -185,7 +191,7 @@ async def _fetch_text(client: httpx.AsyncClient, url: str) -> dict:
|
||||
response = await client.get(url)
|
||||
body = response.text[:MAX_RAW_BYTES]
|
||||
return {"status": response.status_code, "text": body, "url": str(response.url)}
|
||||
except httpx.HTTPError as exc:
|
||||
except (httpx.HTTPError, BlockedAddressError) as exc:
|
||||
return {"status": 0, "text": "", "url": url, "error": str(exc)[:200]}
|
||||
|
||||
|
||||
@@ -256,12 +262,18 @@ async def _site_resources(
|
||||
return robots, sitemap, llms
|
||||
|
||||
|
||||
async def _build_page(context, client, url: str, output_dir, index: int) -> PageContext:
|
||||
async def _build_page(
|
||||
context: Any,
|
||||
client: httpx.AsyncClient,
|
||||
url: str,
|
||||
output_dir: Path | None,
|
||||
index: int,
|
||||
) -> PageContext:
|
||||
page_ctx = PageContext(requested_url=url, url=url)
|
||||
page = await context.new_page()
|
||||
console_errors = []
|
||||
console_errors: list[str] = []
|
||||
|
||||
def _on_console(message):
|
||||
def _on_console(message: Any) -> None:
|
||||
if message.type == "error":
|
||||
console_errors.append(message.text[:200])
|
||||
|
||||
@@ -286,6 +298,8 @@ async def _build_page(context, client, url: str, output_dir, index: int) -> Page
|
||||
chain.append({"url": req.url, "status": resp.status if resp else 0})
|
||||
req = req.redirected_from
|
||||
page_ctx.redirect_chain = list(reversed(chain))
|
||||
for hop in [*(h["url"] for h in page_ctx.redirect_chain), page_ctx.url]:
|
||||
await guard_public_url(hop)
|
||||
dom = await page.evaluate(EXTRACT_SCRIPT)
|
||||
page_ctx.metrics = dom.pop("metrics", {})
|
||||
page_ctx.metrics["consoleErrors"] = len(console_errors)
|
||||
@@ -315,7 +329,9 @@ async def _build_page(context, client, url: str, output_dir, index: int) -> Page
|
||||
return page_ctx
|
||||
|
||||
|
||||
async def crawl_target(payload: dict, emit, output_dir) -> SiteContext:
|
||||
async def crawl_target(
|
||||
payload: dict, emit: Callable[[dict], None], output_dir: Path | None
|
||||
) -> SiteContext:
|
||||
target = _normalize_url(payload.get("url", ""))
|
||||
mode = payload.get("mode", "url")
|
||||
allow_private = bool(payload.get("allow_private"))
|
||||
@@ -332,7 +348,7 @@ async def crawl_target(payload: dict, emit, output_dir) -> SiteContext:
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async with httpx.AsyncClient(
|
||||
async with guarded_async_client(
|
||||
follow_redirects=True,
|
||||
timeout=RAW_FETCH_TIMEOUT,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
@@ -355,8 +371,6 @@ async def crawl_target(payload: dict, emit, output_dir) -> SiteContext:
|
||||
for url in candidates:
|
||||
host = urlparse(url).netloc
|
||||
if host and host == site.base_host:
|
||||
# Same host as the audited target, already validated by the
|
||||
# top-level guard above; skip the redundant per-URL DNS lookup.
|
||||
safe_urls.append(url)
|
||||
continue
|
||||
try:
|
||||
|
||||
@@ -8,6 +8,7 @@ import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from .checks.base import SiteContext
|
||||
from .checks.registry import compute_score, run_page_checks, run_site_checks
|
||||
from .crawler import crawl_target
|
||||
|
||||
@@ -17,7 +18,7 @@ def _emit(frame: dict) -> None:
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def _trim_site(site) -> dict:
|
||||
def _trim_site(site: SiteContext) -> dict:
|
||||
robots = site.robots or {}
|
||||
sitemap = site.sitemap or {}
|
||||
return {
|
||||
@@ -40,8 +41,8 @@ def _trim_site(site) -> dict:
|
||||
async def _run(payload: dict, output_dir: Path) -> dict:
|
||||
site = await crawl_target(payload, _emit, output_dir)
|
||||
|
||||
all_checks = []
|
||||
pages_summary = []
|
||||
all_checks: list = []
|
||||
pages_summary: list[dict] = []
|
||||
_emit({"type": "stage", "stage": "checks", "message": "Running SEO checks"})
|
||||
for page in site.pages:
|
||||
page_checks = run_page_checks(page, site)
|
||||
|
||||
Reference in New Issue
Block a user