chore: replace python -m agents.validator with hawk in all agent mode instructions

This commit is contained in:
2026-06-14 00:36:10 +00:00
parent 64c8c967e5
commit 1076696dec
85 changed files with 682 additions and 5442 deletions
@@ -6,6 +6,7 @@ from .base import (
HIGH,
MEDIUM,
LOW,
Check,
PageContext,
SiteContext,
ok_check,
@@ -16,7 +17,7 @@ CATEGORY = "mobile_accessibility"
@page_check
def responsive(page: PageContext, site: SiteContext) -> PageContext:
def responsive(page: PageContext, site: SiteContext) -> Check:
mobile = page.mobile or {}
overflow = bool(mobile.get("hasHorizontalOverflow"))
return ok_check(
@@ -33,7 +34,7 @@ def responsive(page: PageContext, site: SiteContext) -> PageContext:
@page_check
def tap_targets(page: PageContext, site: SiteContext):
def tap_targets(page: PageContext, site: SiteContext) -> Check | None:
mobile = page.mobile or {}
small = int(mobile.get("smallTapTargets", 0) or 0)
if "smallTapTargets" not in mobile:
@@ -52,7 +53,7 @@ def tap_targets(page: PageContext, site: SiteContext):
@page_check
def image_alt(page: PageContext, site: SiteContext):
def image_alt(page: PageContext, site: SiteContext) -> Check | None:
images = page.dom.get("images", []) or []
if not images:
return None
@@ -71,7 +72,7 @@ def image_alt(page: PageContext, site: SiteContext):
@page_check
def form_labels(page: PageContext, site: SiteContext):
def form_labels(page: PageContext, site: SiteContext) -> Check | None:
missing = int(page.dom.get("formsMissingLabels", 0) or 0)
if not page.dom.get("formFieldCount"):
return None
@@ -89,7 +90,7 @@ def form_labels(page: PageContext, site: SiteContext):
@page_check
def document_title(page: PageContext, site: SiteContext):
def document_title(page: PageContext, site: SiteContext) -> Check:
return ok_check(
"a11y.document_title",
CATEGORY,
@@ -26,7 +26,7 @@ from . import ai_readiness # noqa: F401
from . import crosspage # noqa: F401
def _collect(result) -> list[Check]:
def _collect(result: object) -> list[Check]:
if result is None:
return []
if isinstance(result, Check):
@@ -5,6 +5,7 @@ from __future__ import annotations
from .base import (
MEDIUM,
LOW,
Check,
PageContext,
SiteContext,
ok_check,
@@ -43,7 +44,7 @@ def security_headers(page: PageContext, site: SiteContext) -> list:
@page_check
def tls_valid(page: PageContext, site: SiteContext):
def tls_valid(page: PageContext, site: SiteContext) -> Check | None:
if "tlsValid" not in page.metrics:
return None
valid = bool(page.metrics.get("tlsValid"))
@@ -3,6 +3,7 @@
from __future__ import annotations
import json
from typing import Iterator
from .base import (
MEDIUM,
@@ -39,7 +40,7 @@ REQUIRED_PROPS = {
}
def _iter_objects(parsed):
def _iter_objects(parsed: object) -> Iterator[dict]:
if isinstance(parsed, list):
for item in parsed:
yield from _iter_objects(item)
@@ -54,7 +55,7 @@ def _iter_objects(parsed):
yield parsed
def _types(node) -> list:
def _types(node: dict) -> list:
raw = node.get("@type")
if isinstance(raw, list):
return [str(item) for item in raw]
+23 -9
View File
@@ -3,12 +3,18 @@
from __future__ import annotations
import asyncio
from pathlib import Path
from typing import Any, Callable
from urllib.parse import urljoin, urlparse
from xml.etree import ElementTree
import httpx
from devplacepy.net_guard import BlockedAddressError, guard_public_url
from devplacepy.net_guard import (
BlockedAddressError,
guard_public_url,
guarded_async_client,
)
from .checks.base import PageContext, SiteContext
USER_AGENT = (
@@ -185,7 +191,7 @@ async def _fetch_text(client: httpx.AsyncClient, url: str) -> dict:
response = await client.get(url)
body = response.text[:MAX_RAW_BYTES]
return {"status": response.status_code, "text": body, "url": str(response.url)}
except httpx.HTTPError as exc:
except (httpx.HTTPError, BlockedAddressError) as exc:
return {"status": 0, "text": "", "url": url, "error": str(exc)[:200]}
@@ -256,12 +262,18 @@ async def _site_resources(
return robots, sitemap, llms
async def _build_page(context, client, url: str, output_dir, index: int) -> PageContext:
async def _build_page(
context: Any,
client: httpx.AsyncClient,
url: str,
output_dir: Path | None,
index: int,
) -> PageContext:
page_ctx = PageContext(requested_url=url, url=url)
page = await context.new_page()
console_errors = []
console_errors: list[str] = []
def _on_console(message):
def _on_console(message: Any) -> None:
if message.type == "error":
console_errors.append(message.text[:200])
@@ -286,6 +298,8 @@ async def _build_page(context, client, url: str, output_dir, index: int) -> Page
chain.append({"url": req.url, "status": resp.status if resp else 0})
req = req.redirected_from
page_ctx.redirect_chain = list(reversed(chain))
for hop in [*(h["url"] for h in page_ctx.redirect_chain), page_ctx.url]:
await guard_public_url(hop)
dom = await page.evaluate(EXTRACT_SCRIPT)
page_ctx.metrics = dom.pop("metrics", {})
page_ctx.metrics["consoleErrors"] = len(console_errors)
@@ -315,7 +329,9 @@ async def _build_page(context, client, url: str, output_dir, index: int) -> Page
return page_ctx
async def crawl_target(payload: dict, emit, output_dir) -> SiteContext:
async def crawl_target(
payload: dict, emit: Callable[[dict], None], output_dir: Path | None
) -> SiteContext:
target = _normalize_url(payload.get("url", ""))
mode = payload.get("mode", "url")
allow_private = bool(payload.get("allow_private"))
@@ -332,7 +348,7 @@ async def crawl_target(payload: dict, emit, output_dir) -> SiteContext:
from playwright.async_api import async_playwright
async with httpx.AsyncClient(
async with guarded_async_client(
follow_redirects=True,
timeout=RAW_FETCH_TIMEOUT,
headers={"User-Agent": USER_AGENT},
@@ -355,8 +371,6 @@ async def crawl_target(payload: dict, emit, output_dir) -> SiteContext:
for url in candidates:
host = urlparse(url).netloc
if host and host == site.base_host:
# Same host as the audited target, already validated by the
# top-level guard above; skip the redundant per-URL DNS lookup.
safe_urls.append(url)
continue
try:
+4 -3
View File
@@ -8,6 +8,7 @@ import sys
from datetime import datetime, timezone
from pathlib import Path
from .checks.base import SiteContext
from .checks.registry import compute_score, run_page_checks, run_site_checks
from .crawler import crawl_target
@@ -17,7 +18,7 @@ def _emit(frame: dict) -> None:
sys.stdout.flush()
def _trim_site(site) -> dict:
def _trim_site(site: SiteContext) -> dict:
robots = site.robots or {}
sitemap = site.sitemap or {}
return {
@@ -40,8 +41,8 @@ def _trim_site(site) -> dict:
async def _run(payload: dict, output_dir: Path) -> dict:
site = await crawl_target(payload, _emit, output_dir)
all_checks = []
pages_summary = []
all_checks: list = []
pages_summary: list[dict] = []
_emit({"type": "stage", "stage": "checks", "message": "Running SEO checks"})
for page in site.pages:
page_checks = run_page_checks(page, site)