feat: add deepsearch research system with CLI prune/clear and database schema

Implement a multi-agent deep web research subsystem including CLI commands for pruning expired jobs and clearing all artifacts, database tables for sessions/messages/URL cache with indexes, config paths for chroma storage, and internal embed URL for vector operations.
This commit is contained in:
2026-06-14 01:34:21 +00:00
parent c7770ee21a
commit 4ae0b0db5d
53 changed files with 3956 additions and 18 deletions
@@ -0,0 +1 @@
# retoor <retoor@molodetz.nl>
+107
View File
@@ -0,0 +1,107 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import logging
from dataclasses import dataclass, field
from .embeddings import embed_texts, local_embed
from .llm import complete_chat
from .store import Chunk, VectorStore
logger = logging.getLogger(__name__)
CHAT_TOP_K = 8
MAX_CONTEXT_CHARS = 9000
CHAT_MAX_TOKENS = 900
SYSTEM_PROMPT = (
"You are the DeepSearch research assistant. Answer the user's question using ONLY "
"the numbered SOURCES below, which were gathered during a web research session. "
"Never use outside knowledge or guess. If the sources do not contain the answer, "
"say so plainly. Cite every claim inline with the bracket marker of the source it "
"comes from, like [1] or [2]. Keep the answer focused and well structured in "
"markdown."
)
@dataclass
class ChatAnswer:
text: str
citations: list[dict] = field(default_factory=list)
def _build_context(chunks: list[Chunk]) -> tuple[str, list[dict]]:
blocks: list[str] = []
citations: list[dict] = []
used = 0
for index, chunk in enumerate(chunks, start=1):
snippet = chunk.text.strip()
if not snippet:
continue
header = f"[{index}] {chunk.title or chunk.url} ({chunk.url})"
block = f"{header}\n{snippet}"
if used + len(block) > MAX_CONTEXT_CHARS and blocks:
break
used += len(block)
blocks.append(block)
citations.append(
{
"index": index,
"url": chunk.url,
"title": chunk.title or chunk.url,
"score": round(chunk.score, 4),
}
)
return "\n\n".join(blocks), citations
class DeepsearchChat:
def __init__(self, collection_name: str, api_key: str) -> None:
self.store = VectorStore(collection_name)
self.api_key = api_key
async def _embed_query(self, question: str) -> list[float]:
result = await embed_texts([question], self.api_key)
if not result.vectors:
result = local_embed([question])
return result.vectors[0]
async def retrieve(self, question: str) -> list[Chunk]:
query_vector = await self._embed_query(question)
return self.store.hybrid_search(question, query_vector, top_k=CHAT_TOP_K)
async def answer(self, question: str, history: list[dict] | None = None) -> ChatAnswer:
chunks = await self.retrieve(question)
if not chunks:
return ChatAnswer(
text=(
"The research session did not capture anything relevant to that "
"question. Try rephrasing or running a deeper search."
),
citations=[],
)
context, citations = _build_context(chunks)
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
for turn in (history or [])[-6:]:
role = turn.get("role")
content = turn.get("content")
if role in ("user", "assistant") and content:
messages.append({"role": role, "content": content})
messages.append(
{
"role": "user",
"content": f"SOURCES:\n{context}\n\nQUESTION: {question}",
}
)
try:
text = await complete_chat(
messages, self.api_key, max_tokens=CHAT_MAX_TOKENS
)
except Exception as exc:
logger.warning("deepsearch chat synthesis failed: %s", exc)
text = (
"I could not reach the language model to synthesise an answer, but the "
"most relevant sources are listed below."
)
return ChatAnswer(text=text, citations=citations)
@@ -0,0 +1,71 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import hashlib
import logging
import math
import re
from dataclasses import dataclass
import httpx
from devplacepy.config import INTERNAL_EMBED_MODEL, INTERNAL_EMBED_URL
logger = logging.getLogger(__name__)
EMBED_TIMEOUT_SECONDS = 60.0
LOCAL_EMBED_DIMS = 256
TOKEN_PATTERN = re.compile(r"[a-z0-9]+")
@dataclass
class EmbedResult:
vectors: list[list[float]]
backend: str
def _local_vector(text: str) -> list[float]:
bucket = [0.0] * LOCAL_EMBED_DIMS
tokens = TOKEN_PATTERN.findall((text or "").lower())
if not tokens:
return bucket
for token in tokens:
digest = hashlib.sha1(token.encode("utf-8")).digest()
index = int.from_bytes(digest[:4], "big") % LOCAL_EMBED_DIMS
sign = 1.0 if digest[4] % 2 == 0 else -1.0
bucket[index] += sign
norm = math.sqrt(sum(value * value for value in bucket))
if norm == 0.0:
return bucket
return [value / norm for value in bucket]
def local_embed(texts: list[str]) -> EmbedResult:
return EmbedResult(vectors=[_local_vector(text) for text in texts], backend="local")
async def embed_texts(
texts: list[str], api_key: str, *, gateway_url: str = INTERNAL_EMBED_URL
) -> EmbedResult:
if not texts:
return EmbedResult(vectors=[], backend="empty")
headers = {
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
}
payload = {"model": INTERNAL_EMBED_MODEL, "input": texts}
try:
async with httpx.AsyncClient(timeout=EMBED_TIMEOUT_SECONDS) as client:
response = await client.post(gateway_url, json=payload, headers=headers)
if response.status_code >= 400:
raise RuntimeError(f"embed gateway returned {response.status_code}")
data = response.json()
rows = data.get("data") or []
vectors = [row.get("embedding") or [] for row in rows]
if len(vectors) != len(texts) or any(not vector for vector in vectors):
raise RuntimeError("embed gateway returned an incomplete response")
return EmbedResult(vectors=vectors, backend="gateway")
except Exception as exc:
logger.warning("deepsearch embedding gateway failed, using local: %s", exc)
return local_embed(texts)
+132
View File
@@ -0,0 +1,132 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import html
import json
import logging
from datetime import datetime, timezone
logger = logging.getLogger(__name__)
def _findings(report: dict) -> list[dict]:
return report.get("findings") or []
def _sources(report: dict) -> list[dict]:
return report.get("sources") or []
def _gaps(report: dict) -> list[str]:
return report.get("gaps") or []
def to_markdown(report: dict) -> str:
query = report.get("query", "")
lines: list[str] = [f"# DeepSearch report: {query}", ""]
lines.append(
f"- Score: {report.get('score', 0)} "
f"Confidence: {report.get('confidence', 0)} "
f"Source diversity: {report.get('source_diversity', 0)}"
)
lines.append(
f"- Pages crawled: {report.get('page_count', 0)} "
f"Chunks indexed: {report.get('chunk_count', 0)}"
)
generated = report.get("generated_at") or datetime.now(timezone.utc).isoformat()
lines.append(f"- Generated: {generated}")
lines.append("")
summary = report.get("summary", "")
if summary:
lines.extend(["## Summary", "", summary, ""])
findings = _findings(report)
if findings:
lines.append("## Findings")
lines.append("")
for finding in findings:
title = finding.get("title", "")
detail = finding.get("detail", "")
confidence = finding.get("confidence", 0)
lines.append(f"### {title}")
lines.append("")
lines.append(detail)
citations = finding.get("citations") or []
if citations:
lines.append("")
lines.append("Sources: " + ", ".join(str(c) for c in citations))
lines.append(f"\nConfidence: {confidence}")
lines.append("")
gaps = _gaps(report)
if gaps:
lines.append("## Open gaps")
lines.append("")
for gap in gaps:
lines.append(f"- {gap}")
lines.append("")
sources = _sources(report)
if sources:
lines.append("## Sources")
lines.append("")
for index, source in enumerate(sources, start=1):
title = source.get("title") or source.get("url", "")
url = source.get("url", "")
lines.append(f"{index}. [{title}]({url})")
lines.append("")
return "\n".join(lines)
def to_json(report: dict) -> str:
return json.dumps(report, ensure_ascii=False, indent=2)
def _html_document(report: dict) -> str:
query = html.escape(report.get("query", ""))
parts: list[str] = [
"<html><head><meta charset='utf-8'><style>",
"body{font-family:Arial,Helvetica,sans-serif;color:#1b2330;margin:40px;}",
"h1{font-size:22px;}h2{font-size:17px;margin-top:24px;}h3{font-size:14px;}",
".meta{color:#566;font-size:12px;}a{color:#2d6cdf;}",
"</style></head><body>",
f"<h1>DeepSearch report: {query}</h1>",
"<p class='meta'>"
f"Score {report.get('score', 0)} | Confidence {report.get('confidence', 0)} | "
f"Diversity {report.get('source_diversity', 0)} | "
f"Pages {report.get('page_count', 0)} | Chunks {report.get('chunk_count', 0)}"
"</p>",
]
summary = report.get("summary", "")
if summary:
parts.append("<h2>Summary</h2>")
parts.append(f"<p>{html.escape(summary)}</p>")
findings = _findings(report)
if findings:
parts.append("<h2>Findings</h2>")
for finding in findings:
parts.append(f"<h3>{html.escape(finding.get('title', ''))}</h3>")
parts.append(f"<p>{html.escape(finding.get('detail', ''))}</p>")
parts.append(
f"<p class='meta'>Confidence {finding.get('confidence', 0)}</p>"
)
gaps = _gaps(report)
if gaps:
parts.append("<h2>Open gaps</h2><ul>")
for gap in gaps:
parts.append(f"<li>{html.escape(gap)}</li>")
parts.append("</ul>")
sources = _sources(report)
if sources:
parts.append("<h2>Sources</h2><ol>")
for source in sources:
url = html.escape(source.get("url", ""))
title = html.escape(source.get("title") or source.get("url", ""))
parts.append(f"<li><a href='{url}'>{title}</a></li>")
parts.append("</ol>")
parts.append("</body></html>")
return "".join(parts)
def to_pdf(report: dict) -> bytes:
from weasyprint import HTML
return HTML(string=_html_document(report)).write_pdf()
+44
View File
@@ -0,0 +1,44 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import logging
import httpx
from devplacepy.config import INTERNAL_GATEWAY_URL, INTERNAL_MODEL
logger = logging.getLogger(__name__)
CHAT_TIMEOUT_SECONDS = 120.0
DEFAULT_MAX_TOKENS = 1200
async def complete_chat(
messages: list[dict],
api_key: str,
*,
gateway_url: str = INTERNAL_GATEWAY_URL,
model: str = INTERNAL_MODEL,
max_tokens: int = DEFAULT_MAX_TOKENS,
temperature: float = 0.2,
) -> str:
payload = {
"model": model,
"messages": messages,
"max_tokens": max_tokens,
"temperature": temperature,
}
headers = {
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
}
async with httpx.AsyncClient(timeout=CHAT_TIMEOUT_SECONDS) as client:
response = await client.post(gateway_url, json=payload, headers=headers)
if response.status_code >= 400:
raise RuntimeError(f"chat gateway returned {response.status_code}")
data = response.json()
choices = data.get("choices") or []
if not choices:
raise RuntimeError("chat gateway returned no choices")
return (choices[0].get("message", {}).get("content") or "").strip()
+209
View File
@@ -0,0 +1,209 @@
# retoor <retoor@molodetz.nl>
from __future__ import annotations
import logging
import math
import re
from collections import Counter
from dataclasses import dataclass, field
from devplacepy.config import DEEPSEARCH_CHROMA_DIR
logger = logging.getLogger(__name__)
TOKEN_PATTERN = re.compile(r"[a-z0-9]+")
BM25_K1 = 1.5
BM25_B = 0.75
HYBRID_VECTOR_WEIGHT = 0.6
HYBRID_KEYWORD_WEIGHT = 0.4
DEFAULT_TOP_K = 8
CANDIDATE_MULTIPLIER = 4
@dataclass
class Chunk:
uid: str
text: str
url: str
title: str
depth: int = 0
source: str = ""
position: int = 0
score: float = 0.0
metadata: dict = field(default_factory=dict)
def _tokenize(text: str) -> list[str]:
return TOKEN_PATTERN.findall((text or "").lower())
class VectorStore:
def __init__(self, collection_name: str) -> None:
self.collection_name = collection_name
self._client = None
self._collection = None
def _ensure(self):
if self._collection is not None:
return self._collection
import chromadb
DEEPSEARCH_CHROMA_DIR.mkdir(parents=True, exist_ok=True)
self._client = chromadb.PersistentClient(path=str(DEEPSEARCH_CHROMA_DIR))
self._collection = self._client.get_or_create_collection(
name=self.collection_name, metadata={"hnsw:space": "cosine"}
)
return self._collection
def add(self, chunks: list[Chunk], vectors: list[list[float]]) -> None:
if not chunks:
return
collection = self._ensure()
collection.add(
ids=[chunk.uid for chunk in chunks],
embeddings=vectors,
documents=[chunk.text for chunk in chunks],
metadatas=[
{
"url": chunk.url,
"title": chunk.title,
"depth": chunk.depth,
"source": chunk.source,
"position": chunk.position,
}
for chunk in chunks
],
)
def all_chunks(self) -> list[Chunk]:
collection = self._ensure()
data = collection.get(include=["documents", "metadatas"])
chunks: list[Chunk] = []
ids = data.get("ids") or []
documents = data.get("documents") or []
metadatas = data.get("metadatas") or []
for index, uid in enumerate(ids):
meta = metadatas[index] if index < len(metadatas) else {}
chunks.append(
Chunk(
uid=uid,
text=documents[index] if index < len(documents) else "",
url=str(meta.get("url", "")),
title=str(meta.get("title", "")),
depth=int(meta.get("depth", 0) or 0),
source=str(meta.get("source", "")),
position=int(meta.get("position", 0) or 0),
metadata=dict(meta),
)
)
return chunks
def count(self) -> int:
try:
return self._ensure().count()
except Exception:
return 0
def vector_search(
self, query_vector: list[float], top_k: int, where: dict | None = None
) -> list[Chunk]:
collection = self._ensure()
result = collection.query(
query_embeddings=[query_vector],
n_results=top_k,
where=where or None,
include=["documents", "metadatas", "distances"],
)
ids = (result.get("ids") or [[]])[0]
documents = (result.get("documents") or [[]])[0]
metadatas = (result.get("metadatas") or [[]])[0]
distances = (result.get("distances") or [[]])[0]
chunks: list[Chunk] = []
for index, uid in enumerate(ids):
meta = metadatas[index] if index < len(metadatas) else {}
distance = distances[index] if index < len(distances) else 1.0
chunks.append(
Chunk(
uid=uid,
text=documents[index] if index < len(documents) else "",
url=str(meta.get("url", "")),
title=str(meta.get("title", "")),
depth=int(meta.get("depth", 0) or 0),
source=str(meta.get("source", "")),
position=int(meta.get("position", 0) or 0),
score=1.0 - float(distance),
metadata=dict(meta),
)
)
return chunks
def keyword_scores(self, query: str, chunks: list[Chunk]) -> dict[str, float]:
terms = _tokenize(query)
if not terms or not chunks:
return {}
docs = [_tokenize(chunk.text) for chunk in chunks]
lengths = [len(doc) for doc in docs]
avg_len = (sum(lengths) / len(lengths)) if lengths else 0.0
doc_freq: Counter = Counter()
for doc in docs:
for term in set(doc):
if term in terms:
doc_freq[term] += 1
total_docs = len(docs)
scores: dict[str, float] = {}
for index, chunk in enumerate(chunks):
counts = Counter(docs[index])
length = lengths[index] or 1
score = 0.0
for term in terms:
freq = counts.get(term, 0)
if freq == 0:
continue
idf = math.log(
1 + (total_docs - doc_freq[term] + 0.5) / (doc_freq[term] + 0.5)
)
denom = freq + BM25_K1 * (
1 - BM25_B + BM25_B * (length / (avg_len or 1))
)
score += idf * (freq * (BM25_K1 + 1)) / (denom or 1)
scores[chunk.uid] = score
return scores
def hybrid_search(
self,
query: str,
query_vector: list[float],
top_k: int = DEFAULT_TOP_K,
where: dict | None = None,
) -> list[Chunk]:
candidates = self.vector_search(
query_vector, top_k * CANDIDATE_MULTIPLIER, where
)
if not candidates:
return []
keyword = self.keyword_scores(query, candidates)
vec_max = max((chunk.score for chunk in candidates), default=0.0) or 1.0
kw_max = max(keyword.values(), default=0.0) or 1.0
for chunk in candidates:
vec_norm = max(0.0, chunk.score) / vec_max
kw_norm = keyword.get(chunk.uid, 0.0) / kw_max
chunk.score = (
HYBRID_VECTOR_WEIGHT * vec_norm + HYBRID_KEYWORD_WEIGHT * kw_norm
)
candidates.sort(key=lambda chunk: chunk.score, reverse=True)
return candidates[:top_k]
def drop(self) -> None:
try:
import chromadb
DEEPSEARCH_CHROMA_DIR.mkdir(parents=True, exist_ok=True)
client = self._client or chromadb.PersistentClient(
path=str(DEEPSEARCH_CHROMA_DIR)
)
client.delete_collection(self.collection_name)
except Exception as exc:
logger.info("deepsearch collection drop skipped for %s: %s", self.collection_name, exc)
finally:
self._collection = None