Compare commits

..
Author SHA1 Message Date
Typosaurus 38c2bab503 ticket #111 attempt 1 2026-07-23 03:03:02 +00:00
8 changed files with 195 additions and 121 deletions
File diff suppressed because one or more lines are too long
+2
View File
@@ -38,6 +38,7 @@ def field(
example="", example="",
description="", description="",
options=None, options=None,
nullable=False,
): ):
spec = { spec = {
"name": name, "name": name,
@@ -46,6 +47,7 @@ def field(
"required": required, "required": required,
"example": example, "example": example,
"description": description, "description": description,
"nullable": nullable,
} }
if options: if options:
spec["options"] = list(options) spec["options"] = list(options)
+11 -2
View File
@@ -39,7 +39,15 @@ def _unwrap_optional(annotation):
return annotation return annotation
def _value(name, annotation, stack): def _is_optional(annotation):
if get_origin(annotation) is Union:
return type(None) in get_args(annotation)
return False
def _value(name, annotation, stack, is_nullable=False):
if is_nullable:
return None
annotation = _unwrap_optional(annotation) annotation = _unwrap_optional(annotation)
origin = get_origin(annotation) origin = get_origin(annotation)
if origin in (list, set, tuple): if origin in (list, set, tuple):
@@ -71,7 +79,8 @@ def _from_model(model, stack):
stack = stack | {model} stack = stack | {model}
example = {} example = {}
for name, info in model.model_fields.items(): for name, info in model.model_fields.items():
example[name] = _value(name, info.annotation, stack) is_nullable = _is_optional(info.annotation)
example[name] = _value(name, info.annotation, stack, is_nullable)
return example return example
+111 -114
View File
@@ -22,8 +22,6 @@ from .pdf import MAX_PDF_BYTES, extract_pdf_text, is_pdf
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
_pw_lock = asyncio.Lock()
RSEARCH_URL = "https://rsearch.app.molodetz.nl" RSEARCH_URL = "https://rsearch.app.molodetz.nl"
RSEARCH_TIMEOUT_SECONDS = 45.0 RSEARCH_TIMEOUT_SECONDS = 45.0
FETCH_TIMEOUT_SECONDS = 20.0 FETCH_TIMEOUT_SECONDS = 20.0
@@ -174,7 +172,13 @@ async def _render_with_playwright(
) -> tuple[str, str, int, list[tuple[str, str]]]: ) -> tuple[str, str, int, list[tuple[str, str]]]:
from playwright.async_api import async_playwright from playwright.async_api import async_playwright
async def _render(browser) -> tuple[str, str, int, list[tuple[str, str]]]: own_browser = browser is None
if own_browser:
pw = await async_playwright().__aenter__()
browser = await pw.chromium.launch(
headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"]
)
try:
context = await browser.new_context(user_agent=USER_AGENT) context = await browser.new_context(user_agent=USER_AGENT)
page = await context.new_page() page = await context.new_page()
response = await page.goto(url, wait_until="load", timeout=30000) response = await page.goto(url, wait_until="load", timeout=30000)
@@ -185,18 +189,10 @@ async def _render_with_playwright(
await context.close() await context.close()
extracted = extract_html(content, base_url=url) extracted = extract_html(content, base_url=url)
return extracted.title, extracted.text, status, extracted.links return extracted.title, extracted.text, status, extracted.links
finally:
if browser is None: if own_browser:
async with async_playwright() as pw: await browser.close()
browser = await pw.chromium.launch( await pw.__aexit__(None, None, None)
headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"]
)
try:
return await _render(browser)
finally:
await browser.close()
else:
return await _render(browser)
async def fetch_page(url: str, depth: int, browser=None) -> CrawledPage | None: async def fetch_page(url: str, depth: int, browser=None) -> CrawledPage | None:
@@ -246,20 +242,18 @@ async def fetch_page(url: str, depth: int, browser=None) -> CrawledPage | None:
except (LookupError, ValueError) as exc: except (LookupError, ValueError) as exc:
logger.info("deepsearch decode failed for %s: %s", url, exc) logger.info("deepsearch decode failed for %s: %s", url, exc)
if len(text) < MIN_PAGE_CHARS: if len(text) < MIN_PAGE_CHARS:
async with _pw_lock: try:
try: r_title, r_text, r_status, r_links = await _render_with_playwright(url, browser)
r_title, r_text, r_status, r_links = await _render_with_playwright(url, browser) if len(r_text) > len(text):
except Exception as exc: title, text, status, source, links = (
logger.info("deepsearch render failed for %s: %s", url, exc) r_title or title,
r_title, r_text, r_status, r_links = "", "", 0, [] r_text,
if len(r_text) > len(text): r_status or status,
title, text, status, source, links = ( "playwright",
r_title or title, r_links,
r_text, )
r_status or status, except Exception as exc:
"playwright", logger.info("deepsearch render failed for %s: %s", url, exc)
r_links,
)
if len(text) < MIN_PAGE_CHARS: if len(text) < MIN_PAGE_CHARS:
return None return None
return CrawledPage( return CrawledPage(
@@ -302,99 +296,102 @@ async def crawl(
total = min(len(level_candidates), max_pages) total = min(len(level_candidates), max_pages)
cancelled = False cancelled = False
pw = None
browser = None browser = None
async with async_playwright() as pw: try:
try: pw = await async_playwright().__aenter__()
browser = await pw.chromium.launch( browser = await pw.chromium.launch(
headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"] headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"]
) )
except Exception as exc: except Exception as exc:
logger.warning("deepsearch playwright launch failed, pages will use httpx only: %s", exc) logger.warning("deepsearch playwright launch failed, pages will use httpx only: %s", exc)
try: try:
for level in range(max(1, depth)): for level in range(max(1, depth)):
if cancelled or fetched >= max_pages or not level_candidates: if cancelled or fetched >= max_pages or not level_candidates:
break
next_candidates: list[dict] = []
for start in range(0, len(level_candidates), CRAWL_CONCURRENCY):
if fetched >= max_pages:
break break
next_candidates: list[dict] = [] if await should_stop():
for start in range(0, len(level_candidates), CRAWL_CONCURRENCY): emit({"type": "stage", "stage": "cancelled", "message": "Crawl cancelled"})
cancelled = True
break
batch = level_candidates[start : start + CRAWL_CONCURRENCY][: max_pages - fetched]
for candidate in batch:
emit(
{
"type": "progress",
"done": fetched,
"total": total,
"url": candidate["url"],
"depth": level,
"message": f"Reading {candidate['url']}",
}
)
if is_cached(candidate["url"]):
emit({"type": "page_cached", "url": candidate["url"], "reason": "seen in a prior run"})
fetch_start = time.perf_counter()
results = await asyncio.gather(
*(_resolve_candidate(candidate, level, browser) for candidate in batch),
return_exceptions=True,
)
elapsed_ms = int((time.perf_counter() - fetch_start) * 1000)
for candidate, page in zip(batch, results):
url = candidate["url"]
if isinstance(page, BaseException):
logger.info("deepsearch fetch crashed for %s: %s", url, page)
page = None
if page is None:
emit(
{
"type": "page_skipped",
"url": url,
"reason": "no readable content",
"elapsed_ms": elapsed_ms,
}
)
continue
if fetched >= max_pages: if fetched >= max_pages:
break break
if await should_stop(): digest = content_hash(page.text)
emit({"type": "stage", "stage": "cancelled", "message": "Crawl cancelled"}) if digest in outcome.seen_hashes:
cancelled = True
break
batch = level_candidates[start : start + CRAWL_CONCURRENCY][: max_pages - fetched]
for candidate in batch:
emit( emit(
{ {
"type": "progress", "type": "page_duplicate",
"done": fetched, "url": url,
"total": total, "reason": "duplicate content",
"url": candidate["url"],
"depth": level,
"message": f"Reading {candidate['url']}",
}
)
if is_cached(candidate["url"]):
emit({"type": "page_cached", "url": candidate["url"], "reason": "seen in a prior run"})
fetch_start = time.perf_counter()
results = await asyncio.gather(
*(_resolve_candidate(candidate, level, browser) for candidate in batch),
return_exceptions=True,
)
elapsed_ms = int((time.perf_counter() - fetch_start) * 1000)
for candidate, page in zip(batch, results):
url = candidate["url"]
if isinstance(page, BaseException):
logger.info("deepsearch fetch crashed for %s: %s", url, page)
page = None
if page is None:
emit(
{
"type": "page_skipped",
"url": url,
"reason": "no readable content",
"elapsed_ms": elapsed_ms,
}
)
continue
if fetched >= max_pages:
break
digest = content_hash(page.text)
if digest in outcome.seen_hashes:
emit(
{
"type": "page_duplicate",
"url": url,
"reason": "duplicate content",
"elapsed_ms": elapsed_ms,
}
)
continue
outcome.seen_hashes.add(digest)
outcome.pages.append(page)
fetched += 1
emit(
{
"type": "page_loaded",
"url": page.url,
"title": page.title,
"source": page.source,
"depth": level,
"render": page.source == "playwright",
"elapsed_ms": elapsed_ms, "elapsed_ms": elapsed_ms,
"done": fetched,
"total": total,
} }
) )
if level + 1 < depth: continue
for link in relevant_links(page.links, query, LINKS_PER_PAGE): outcome.seen_hashes.add(digest)
if link not in seen_urls: outcome.pages.append(page)
seen_urls.add(link) fetched += 1
next_candidates.append({"url": link}) emit(
level_candidates = next_candidates {
total = min(total + len(next_candidates), max_pages) "type": "page_loaded",
finally: "url": page.url,
if browser is not None: "title": page.title,
await browser.close() "source": page.source,
"depth": level,
"render": page.source == "playwright",
"elapsed_ms": elapsed_ms,
"done": fetched,
"total": total,
}
)
if level + 1 < depth:
for link in relevant_links(page.links, query, LINKS_PER_PAGE):
if link not in seen_urls:
seen_urls.add(link)
next_candidates.append({"url": link})
level_candidates = next_candidates
total = min(total + len(next_candidates), max_pages)
finally:
if browser is not None:
await browser.close()
if pw is not None:
await pw.__aexit__(None, None, None)
return outcome return outcome
-4
View File
@@ -620,10 +620,6 @@ img {
.topnav-logo span { color: var(--accent); } .topnav-logo span { color: var(--accent); }
.topnav-links { display: flex; gap: 0.25rem; } .topnav-links { display: flex; gap: 0.25rem; }
.topnav-link { .topnav-link {
display: inline-flex;
align-items: center;
gap: 0.375rem;
white-space: nowrap;
padding: 0.5rem 0.75rem; padding: 0.5rem 0.75rem;
border-radius: var(--radius); border-radius: var(--radius);
font-size: 0.875rem; font-size: 0.875rem;
+10
View File
@@ -218,6 +218,16 @@
font-weight: 700; font-weight: 700;
} }
.param-nullable {
font-size: 0.625rem;
text-transform: uppercase;
letter-spacing: 0.05em;
color: var(--text-muted);
border: 1px dashed var(--border);
border-radius: 999px;
padding: 0.05rem 0.4rem;
}
.param-loc { .param-loc {
font-size: 0.625rem; font-size: 0.625rem;
text-transform: uppercase; text-transform: uppercase;
+1
View File
@@ -139,6 +139,7 @@ export class ApiTester {
const label = this.el("div", { class: "param-label" }, [ const label = this.el("div", { class: "param-label" }, [
this.el("span", { class: "param-name", text: param.name }), this.el("span", { class: "param-name", text: param.name }),
param.required ? this.el("span", { class: "param-required", text: "*" }) : null, param.required ? this.el("span", { class: "param-required", text: "*" }) : null,
param.nullable ? this.el("span", { class: "param-nullable", text: "nullable" }) : null,
this.el("span", { class: "param-loc param-loc-" + param.location, text: param.location }), this.el("span", { class: "param-loc param-loc-" + param.location, text: param.location }),
]); ]);
const allowed = param.type === "enum" && param.options && param.options.length const allowed = param.type === "enum" && param.options && param.options.length
+59
View File
@@ -0,0 +1,59 @@
# retoor <retoor@molodetz.nl>
from typing import Optional
from pydantic import BaseModel
from devplacepy.docs_api._shared import field
from devplacepy.docs_examples import _is_optional, schema_example
def test_field_nullable_parameter():
f = field("bio", "query", "string", False, "", "User biography", nullable=True)
assert f["nullable"] is True
f2 = field("username", "path", "string", True, "alice", "Target username")
assert f2["nullable"] is False
def test_is_optional_optional_type():
assert _is_optional(Optional[str]) is True
assert _is_optional(Optional[int]) is True
assert _is_optional(Optional[list[str]]) is True
def test_is_optional_non_optional():
assert _is_optional(str) is False
assert _is_optional(int) is False
assert _is_optional(list[str]) is False
assert _is_optional(dict) is False
def test_schema_example_nullable_fields():
class NullableModel(BaseModel):
name: str
bio: Optional[str] = None
age: int
score: Optional[int] = None
tags: Optional[list[str]] = None
result = schema_example(NullableModel)
assert result["name"] == "string"
assert result["age"] == 0
assert result["bio"] is None
assert result["score"] is None
assert result["tags"] is None
def test_schema_example_non_nullable_fields_unaffected():
class StrictModel(BaseModel):
x: str
y: int
z: bool
result = schema_example(StrictModel)
assert result["x"] == "string"
assert result["y"] == 0
assert result["z"] is False