feat: add seo_meta service for AI-generated SEO metadata with CLI management and database layer
DevPlace CI / test (push) Failing after 2m13s

Implement a new `SeoMetaService` subservice that generates clean SEO title/description/keywords for published content items, distinct from the existing SEO diagnostics auditor. Add `seo_metadata` polymorphic table with soft-delete support, batch query methods, and usage tracking. Extend the CLI with `seo-meta prune` and `seo-meta clear` commands for job row lifecycle management. Wire `schedule_seo_meta_for_table` into content creation and editing flows in `content.py`. Document the new service in `AGENTS.md` and `README.md`, including the `extra_head` site setting for custom `<head>` injection.
This commit is contained in:
2026-06-19 20:15:22 +00:00
parent 426d3639c6
commit d10f1af118
51 changed files with 2262 additions and 93 deletions
+23 -1
View File
@@ -3,6 +3,7 @@
from __future__ import annotations
import logging
import re
from dataclasses import dataclass, field
from .embeddings import embed_texts, local_embed
@@ -11,6 +12,8 @@ from .store import Chunk, VectorStore
logger = logging.getLogger(__name__)
CITATION_MARKER = re.compile(r"\[(\d+)\]")
CHAT_TOP_K = 8
MAX_CONTEXT_CHARS = 9000
CHAT_MAX_TOKENS = 900
@@ -63,12 +66,22 @@ class DeepsearchChat:
async def _embed_query(self, question: str) -> list[float]:
result = await embed_texts([question], self.api_key)
if not result.vectors:
if not result.vectors or not result.vectors[0]:
result = local_embed([question])
return result.vectors[0]
async def retrieve(self, question: str) -> list[Chunk]:
query_vector = await self._embed_query(question)
stored_dim = self.store.dims
if stored_dim is not None and len(query_vector) != stored_dim:
query_vector = local_embed([question]).vectors[0]
if len(query_vector) != stored_dim:
logger.warning(
"deepsearch query embedding dim %d != stored %d",
len(query_vector),
stored_dim,
)
return []
return self.store.hybrid_search(question, query_vector, top_k=CHAT_TOP_K)
async def answer(self, question: str, history: list[dict] | None = None) -> ChatAnswer:
@@ -104,4 +117,13 @@ class DeepsearchChat:
"I could not reach the language model to synthesise an answer, but the "
"most relevant sources are listed below."
)
valid = {citation["index"] for citation in citations}
text = self._strip_unmatched_markers(text, valid)
return ChatAnswer(text=text, citations=citations)
def _strip_unmatched_markers(self, text: str, valid: set[int]) -> str:
def replace(match: re.Match) -> str:
index = int(match.group(1))
return match.group(0) if index in valid else ""
return CITATION_MARKER.sub(replace, text)
+14 -2
View File
@@ -16,6 +16,7 @@ logger = logging.getLogger(__name__)
EMBED_TIMEOUT_SECONDS = 60.0
LOCAL_EMBED_DIMS = 256
EMBED_CACHE_MAX = 5000
TOKEN_PATTERN = re.compile(r"[a-z0-9]+")
@@ -26,10 +27,18 @@ class EmbedResult:
latency_ms: int = 0
cache_hits: int = 0
@property
def dims(self) -> int:
for vector in self.vectors:
if vector:
return len(vector)
return 0
@dataclass
class EmbeddingCache:
store: dict[str, list[float]] = field(default_factory=dict)
max_entries: int = EMBED_CACHE_MAX
def key(self, text: str) -> str:
return hashlib.sha1((text or "").encode("utf-8")).hexdigest()
@@ -38,8 +47,11 @@ class EmbeddingCache:
return self.store.get(self.key(text))
def put(self, text: str, vector: list[float]) -> None:
if vector:
self.store[self.key(text)] = vector
if not vector:
return
if len(self.store) >= self.max_entries:
self.store.pop(next(iter(self.store)), None)
self.store[self.key(text)] = vector
def _local_vector(text: str) -> list[float]:
+39 -15
View File
@@ -3,6 +3,7 @@
from __future__ import annotations
import logging
import time
from devplacepy import stealth
from devplacepy.config import INTERNAL_GATEWAY_URL, INTERNAL_MODEL
@@ -13,6 +14,36 @@ CHAT_TIMEOUT_SECONDS = 120.0
DEFAULT_MAX_TOKENS = 1200
async def request_completion(
messages: list[dict],
api_key: str,
*,
gateway_url: str = INTERNAL_GATEWAY_URL,
model: str = INTERNAL_MODEL,
max_tokens: int = DEFAULT_MAX_TOKENS,
temperature: float = 0.2,
timeout: float = CHAT_TIMEOUT_SECONDS,
) -> tuple[str, dict, int]:
payload = {
"model": model,
"messages": messages,
"max_tokens": max_tokens,
"temperature": temperature,
}
headers = {
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
}
start: float = time.monotonic()
async with stealth.stealth_async_client(timeout=timeout) as client:
response = await client.post(gateway_url, json=payload, headers=headers)
elapsed_ms: int = int((time.monotonic() - start) * 1000)
if response.status_code >= 400:
raise RuntimeError(f"chat gateway returned {response.status_code}")
data: dict = response.json()
return data, data.get("usage") or {}, elapsed_ms
async def complete_chat(
messages: list[dict],
api_key: str,
@@ -22,21 +53,14 @@ async def complete_chat(
max_tokens: int = DEFAULT_MAX_TOKENS,
temperature: float = 0.2,
) -> str:
payload = {
"model": model,
"messages": messages,
"max_tokens": max_tokens,
"temperature": temperature,
}
headers = {
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
}
async with stealth.stealth_async_client(timeout=CHAT_TIMEOUT_SECONDS) as client:
response = await client.post(gateway_url, json=payload, headers=headers)
if response.status_code >= 400:
raise RuntimeError(f"chat gateway returned {response.status_code}")
data = response.json()
data, _usage, _elapsed_ms = await request_completion(
messages,
api_key,
gateway_url=gateway_url,
model=model,
max_tokens=max_tokens,
temperature=temperature,
)
choices = data.get("choices") or []
if not choices:
raise RuntimeError("chat gateway returned no choices")
+35
View File
@@ -42,6 +42,7 @@ class VectorStore:
self.collection_name = collection_name
self._client = None
self._collection = None
self._dims: int | None = None
def _ensure(self):
if self._collection is not None:
@@ -55,9 +56,43 @@ class VectorStore:
)
return self._collection
@property
def dims(self) -> int | None:
if self._dims is not None:
return self._dims
try:
collection = self._ensure()
data = collection.get(include=["embeddings"], limit=1)
rows = data.get("embeddings") or []
if rows and rows[0]:
self._dims = len(rows[0])
except Exception:
return self._dims
return self._dims
def add(self, chunks: list[Chunk], vectors: list[list[float]]) -> None:
if not chunks:
return
keep_chunks: list[Chunk] = []
keep_vectors: list[list[float]] = []
for chunk, vector in zip(chunks, vectors):
if not vector:
continue
if self._dims is None:
self._dims = len(vector)
if len(vector) != self._dims:
logger.warning(
"deepsearch dropping chunk with mismatched embedding dim %d != %d",
len(vector),
self._dims,
)
continue
keep_chunks.append(chunk)
keep_vectors.append(vector)
if not keep_chunks:
return
chunks = keep_chunks
vectors = keep_vectors
collection = self._ensure()
collection.add(
ids=[chunk.uid for chunk in chunks],