forked from retoor/devplacepy
feat: add featured/locked columns and auto-rotation logic to news pipeline
- Add `ai_grade`, `featured`, `featured_locked`, `landing_locked`, `author`, `article_published`, `image_url`, `has_unique_image` columns to news table - Extend `news_images` schema with `alt_text`, `phash`, `width`, `height`, `is_placeholder` columns - Create `idx_news_featured` index for efficient featured queries - Update admin toggle endpoints to set `featured_locked`/`landing_locked` when manually toggling - Propagate `featured` and `image_url` fields through landing page, news list, and detail page rendering - Update `get_featured_news` to return `featured` and `image_url` fields - Document in AGENTS.md the full zero-maintenance pipeline: image perceptual hashing, placeholder detection, AI grading on cleaned text, reliability gate, effective score computation, and post-loop landing rotation - Update README.md service description to reflect automatic image comparison and landing rotation capabilities - Clarify docs_api.py summaries that toggling featured/landing now locks articles from auto-rotation
This commit is contained in:
@@ -120,6 +120,7 @@ class BaseService(ABC):
|
||||
default_enabled = True
|
||||
title = ""
|
||||
description = ""
|
||||
details = ""
|
||||
DEFAULT_LOG_SIZE = 20
|
||||
TICK_SECONDS = 1
|
||||
PERSIST_SECONDS = 3
|
||||
@@ -414,6 +415,7 @@ class BaseService(ABC):
|
||||
"name": self.name,
|
||||
"title": self.title or self.name.capitalize(),
|
||||
"description": self.description,
|
||||
"details": self.details,
|
||||
"enabled": enabled,
|
||||
"status": status,
|
||||
"interval_seconds": self.current_interval(),
|
||||
|
||||
+610
-184
@@ -1,13 +1,19 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
import asyncio
|
||||
import io
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from datetime import datetime, timezone
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import httpx
|
||||
import imagehash
|
||||
from PIL import Image
|
||||
|
||||
from devplacepy import stealth
|
||||
from devplacepy import net_guard
|
||||
from devplacepy.config import INTERNAL_GATEWAY_URL, INTERNAL_MODEL
|
||||
from devplacepy.database import get_table, get_setting, internal_gateway_key
|
||||
from devplacepy.services.base import BaseService, ConfigField
|
||||
@@ -22,6 +28,138 @@ AI_MODEL_DEFAULT = INTERNAL_MODEL
|
||||
GRADE_THRESHOLD_DEFAULT = 7
|
||||
GRADE_MAX_TOKENS = 2000
|
||||
|
||||
MIN_TITLE_CHARS = 12
|
||||
MIN_BODY_CHARS = 200
|
||||
MAX_TITLE_CAPS_RATIO = 0.6
|
||||
MIN_IMAGE_DIMENSION = 100
|
||||
PHASH_DISTANCE = 5
|
||||
IMG_PER_ARTICLE = 5
|
||||
IMG_FETCH_MAX_BYTES = 5_000_000
|
||||
IMG_FETCH_TIMEOUT = 8.0
|
||||
UNIQUE_IMAGE_BONUS = 2
|
||||
THIN_CONTENT_PENALTY = 2
|
||||
FEATURE_MIN_SCORE = 8
|
||||
LANDING_MIN_SCORE = 9
|
||||
LANDING_MAX = 6
|
||||
LANDING_RECENCY_DAYS = 4
|
||||
|
||||
JUNK_PATTERNS = (
|
||||
re.compile(r"submitted by\s+/?u/\S+(?:\s+to\s+/?r/\S+)?", re.IGNORECASE),
|
||||
re.compile(r"submitted by\s+\S+\s+to\s+/?r/\S+", re.IGNORECASE),
|
||||
re.compile(r"\[\s*link\s*\]", re.IGNORECASE),
|
||||
re.compile(r"\[\s*\d*\s*comments?\s*\]", re.IGNORECASE),
|
||||
re.compile(r"submitted by[^\[]*\[link\]\s*\[comments?\]", re.IGNORECASE),
|
||||
)
|
||||
WHITESPACE_PATTERN = re.compile(r"\s+")
|
||||
|
||||
NEWS_SUMMARY = (
|
||||
"Fully automatic, zero-maintenance news pipeline: fetches the feed, cleans "
|
||||
"each article, fetches and perceptually compares images to reject "
|
||||
"placeholders, grades deterministically, and auto-rotates Featured and "
|
||||
"Landing. The full grading algorithm is detailed below."
|
||||
)
|
||||
|
||||
GRADE_PROMPT_SPEC = (
|
||||
"You are an expert content strategist and editor for DevPlace, a "
|
||||
"server-rendered social network for software developers. Decide how "
|
||||
"strongly a given article should be published, expressed as a single "
|
||||
"grade.\n\n"
|
||||
"Site context:\n"
|
||||
"- Audience: software developers, hobbyists and professionals, "
|
||||
"interested in software development, AI agents, open-source tools, "
|
||||
"systems programming, and self-hosting.\n"
|
||||
"- Mission: deliver deep technical insight, critical analysis, and "
|
||||
"practical value; build authority and reader trust; avoid low-effort, "
|
||||
"rehashed, or clickbait content.\n"
|
||||
"- Tone: technical, practical, nuanced, skeptical of hype; no "
|
||||
"sensationalism or unverified claims.\n\n"
|
||||
"Evaluate the article against these weighted criteria and weigh them "
|
||||
"internally:\n"
|
||||
"1. Relevance and alignment to the developer niche and audience needs "
|
||||
"(20%).\n"
|
||||
"2. Quality and accuracy: depth, originality, factual correctness, "
|
||||
"sourcing, technical correctness, clear writing (25%).\n"
|
||||
"3. Engagement and readability: strong title/intro/flow, useful "
|
||||
"examples, appropriate length (15%).\n"
|
||||
"4. Originality and uniqueness: new perspective, data, or analysis; not "
|
||||
"duplicated generic web content (15%).\n"
|
||||
"5. SEO and discoverability: keyword opportunity without stuffing, "
|
||||
"shareable (10%).\n"
|
||||
"6. Ethics, legality and risk: no misinformation, plagiarism, or "
|
||||
"brand-safety issues; controversial topics handled with nuance and "
|
||||
"evidence (10%).\n"
|
||||
"7. Strategic fit for the site (5%).\n\n"
|
||||
"Convert your overall judgment to a single integer grade from 1 to 10:\n"
|
||||
"- 9-10: excellent, publish as-is.\n"
|
||||
"- 7-8: good, worth publishing.\n"
|
||||
"- 5-6: marginal, weak fit or quality.\n"
|
||||
"- 1-4: reject, low quality or off-topic.\n\n"
|
||||
"Return ONLY a single integer from 1 to 10, nothing else."
|
||||
)
|
||||
|
||||
GRADING_RULES_DESCRIPTION = (
|
||||
"Each run fetches the feed, cleans every article, fetches and perceptually "
|
||||
"compares images, grades deterministically, and auto-rotates Featured and "
|
||||
"Landing.\n\n"
|
||||
"Content cleaning: after HTML stripping it removes Reddit boilerplate "
|
||||
"(submitted by /u/<user>, submitted by ... to /r/<sub>, standalone [link], "
|
||||
"[comments], [N comments] and trailing submitted-by [link] [comments]) and "
|
||||
"collapses whitespace, applied to title (light), description and content "
|
||||
"before grading and before storage.\n\n"
|
||||
"Reliability gate (forces draft, never Featured or Landing) when: the url is "
|
||||
f"empty or invalid; the cleaned title is shorter than {MIN_TITLE_CHARS} chars; "
|
||||
f"the combined cleaned body is shorter than {MIN_BODY_CHARS} chars; or the "
|
||||
f"title uppercase-letter ratio exceeds {MAX_TITLE_CAPS_RATIO}.\n\n"
|
||||
"Image uniqueness: up to "
|
||||
f"{IMG_PER_ARTICLE} candidate images per article are fetched (max "
|
||||
f"{IMG_FETCH_MAX_BYTES} bytes), decoded and perceptually hashed (phash). An "
|
||||
f"image under {MIN_IMAGE_DIMENSION}px on either side, or one that fails to "
|
||||
"fetch or decode, is a placeholder. Hashes within Hamming distance "
|
||||
f"{PHASH_DISTANCE} that span two or more different articles are a shared "
|
||||
"logo or placeholder and are rejected for all of them. An article with at "
|
||||
"least one genuinely unique image is marked has_unique_image and its first "
|
||||
"such image becomes the primary image.\n\n"
|
||||
"Grading prompt: the exact rubric is the editable news_grade_prompt field "
|
||||
"(Configuration tab), sent to the model at temperature 0 with the cleaned "
|
||||
"Title, Description and Content appended; it must return a single integer "
|
||||
"from 1 to 10.\n\n"
|
||||
"Scoring: the AI grade (1-10, temperature 0) is the base. effective_score = "
|
||||
f"clamp(ai_grade + {UNIQUE_IMAGE_BONUS} if unique image - "
|
||||
f"{THIN_CONTENT_PENALTY} if the body is marginal but not gated, 1..10). The "
|
||||
"effective score is stored in grade (so the listing reflects final quality) "
|
||||
"and the raw AI grade in ai_grade.\n\n"
|
||||
"Promotion: an article is published when valid and effective_score is at or "
|
||||
"above the grade threshold. It is Featured when published, has a unique "
|
||||
f"image and scores at least {FEATURE_MIN_SCORE}. After the run the service "
|
||||
"rotates Landing: the top "
|
||||
f"{LANDING_MAX} published, featured, unique-image articles from the last "
|
||||
f"{LANDING_RECENCY_DAYS} days scoring at least {LANDING_MIN_SCORE} are shown "
|
||||
"on the landing page and the rest of the service-managed set is cleared. "
|
||||
"Rows an admin has manually toggled are locked (featured_locked or "
|
||||
"landing_locked) and the service never auto-manages them again."
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ImageCandidate:
|
||||
url: str
|
||||
alt_text: str = ""
|
||||
phash: str = ""
|
||||
width: int = 0
|
||||
height: int = 0
|
||||
is_placeholder: bool = True
|
||||
|
||||
|
||||
@dataclass
|
||||
class ArticleGrade:
|
||||
ai_grade: int | None
|
||||
effective_score: int
|
||||
has_unique_image: bool
|
||||
image_url: str
|
||||
valid: bool
|
||||
reject_reason: str
|
||||
candidates: list[ImageCandidate] = field(default_factory=list)
|
||||
|
||||
|
||||
def _get_ai_key() -> str:
|
||||
key = os.environ.get("NEWS_AI_KEY")
|
||||
@@ -42,6 +180,62 @@ def _extract_grade(text: str) -> int | None:
|
||||
return None
|
||||
|
||||
|
||||
def clean_news_text(text: str) -> str:
|
||||
if not text:
|
||||
return ""
|
||||
cleaned = strip_html(text)
|
||||
for pattern in JUNK_PATTERNS:
|
||||
cleaned = pattern.sub(" ", cleaned)
|
||||
cleaned = WHITESPACE_PATTERN.sub(" ", cleaned)
|
||||
return cleaned.strip()
|
||||
|
||||
|
||||
def _caps_ratio(title: str) -> float:
|
||||
letters = [c for c in title if c.isalpha()]
|
||||
if not letters:
|
||||
return 0.0
|
||||
uppers = sum(1 for c in letters if c.isupper())
|
||||
return uppers / len(letters)
|
||||
|
||||
|
||||
def _is_valid_url(url: str) -> bool:
|
||||
return url.startswith("http://") or url.startswith("https://")
|
||||
|
||||
|
||||
def reliability_reason(title: str, body: str, url: str) -> str:
|
||||
if not url or not _is_valid_url(url):
|
||||
return "invalid_url"
|
||||
if len(title) < MIN_TITLE_CHARS:
|
||||
return "short_title"
|
||||
if len(body) < MIN_BODY_CHARS:
|
||||
return "thin_body"
|
||||
if _caps_ratio(title) > MAX_TITLE_CAPS_RATIO:
|
||||
return "shouting_title"
|
||||
return ""
|
||||
|
||||
|
||||
def effective_score(
|
||||
ai_grade: int, has_unique_image: bool, body_marginal: bool
|
||||
) -> int:
|
||||
score = ai_grade
|
||||
if has_unique_image:
|
||||
score += UNIQUE_IMAGE_BONUS
|
||||
if body_marginal:
|
||||
score -= THIN_CONTENT_PENALTY
|
||||
return max(1, min(10, score))
|
||||
|
||||
|
||||
def _decode_phash(data: bytes) -> tuple[str, int, int] | None:
|
||||
try:
|
||||
with Image.open(io.BytesIO(data)) as image:
|
||||
image.load()
|
||||
width, height = image.size
|
||||
phash = str(imagehash.phash(image))
|
||||
return phash, width, height
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
async def _get_article_images(url: str, client: httpx.AsyncClient) -> list[dict]:
|
||||
try:
|
||||
resp = await client.get(url, timeout=10.0)
|
||||
@@ -64,15 +258,70 @@ async def _get_article_images(url: str, client: httpx.AsyncClient) -> list[dict]
|
||||
return []
|
||||
|
||||
|
||||
async def _fetch_image_candidate(
|
||||
src: dict, image_client: httpx.AsyncClient
|
||||
) -> ImageCandidate:
|
||||
candidate = ImageCandidate(url=src["url"], alt_text=src.get("alt_text", ""))
|
||||
try:
|
||||
await net_guard.guard_public_url(candidate.url)
|
||||
except Exception:
|
||||
return candidate
|
||||
try:
|
||||
resp = await image_client.get(candidate.url, timeout=IMG_FETCH_TIMEOUT)
|
||||
resp.raise_for_status()
|
||||
data = resp.content[:IMG_FETCH_MAX_BYTES]
|
||||
except Exception:
|
||||
return candidate
|
||||
if len(data) >= IMG_FETCH_MAX_BYTES:
|
||||
return candidate
|
||||
loop = asyncio.get_running_loop()
|
||||
decoded = await loop.run_in_executor(None, _decode_phash, data)
|
||||
if decoded is None:
|
||||
return candidate
|
||||
phash, width, height = decoded
|
||||
candidate.phash = phash
|
||||
candidate.width = width
|
||||
candidate.height = height
|
||||
candidate.is_placeholder = (
|
||||
width < MIN_IMAGE_DIMENSION or height < MIN_IMAGE_DIMENSION
|
||||
)
|
||||
return candidate
|
||||
|
||||
|
||||
def _flag_shared_placeholders(by_article: dict[str, list[ImageCandidate]]) -> None:
|
||||
entries: list[tuple[str, ImageCandidate, imagehash.ImageHash]] = []
|
||||
for article_uid, candidates in by_article.items():
|
||||
for candidate in candidates:
|
||||
if candidate.is_placeholder or not candidate.phash:
|
||||
continue
|
||||
entries.append(
|
||||
(article_uid, candidate, imagehash.hex_to_hash(candidate.phash))
|
||||
)
|
||||
for i in range(len(entries)):
|
||||
uid_a, cand_a, hash_a = entries[i]
|
||||
for j in range(i + 1, len(entries)):
|
||||
uid_b, cand_b, hash_b = entries[j]
|
||||
if uid_a == uid_b:
|
||||
continue
|
||||
if (hash_a - hash_b) <= PHASH_DISTANCE:
|
||||
cand_a.is_placeholder = True
|
||||
cand_b.is_placeholder = True
|
||||
|
||||
|
||||
def _primary_image(candidates: list[ImageCandidate]) -> str:
|
||||
for candidate in candidates:
|
||||
if not candidate.is_placeholder:
|
||||
return candidate.url
|
||||
return ""
|
||||
|
||||
|
||||
class NewsService(BaseService):
|
||||
interval_key = "news_service_interval"
|
||||
min_interval = 60
|
||||
default_enabled = True
|
||||
title = "News"
|
||||
description = (
|
||||
"Periodically fetches developer news articles from the configured feed, "
|
||||
"grades each one with an AI model, and stores them - auto-publishing those "
|
||||
"at or above the grade threshold and saving the rest as drafts."
|
||||
)
|
||||
description = NEWS_SUMMARY
|
||||
details = GRADING_RULES_DESCRIPTION
|
||||
config_fields = [
|
||||
ConfigField(
|
||||
"news_api_url",
|
||||
@@ -87,7 +336,7 @@ class NewsService(BaseService):
|
||||
"AI grading URL",
|
||||
type="url",
|
||||
default=AI_URL_DEFAULT,
|
||||
help="Chat-completions endpoint used to grade each article.",
|
||||
help="Chat-completions endpoint used to grade each cleaned article.",
|
||||
group="AI grading",
|
||||
),
|
||||
ConfigField(
|
||||
@@ -98,6 +347,19 @@ class NewsService(BaseService):
|
||||
help="Model name sent to the grading endpoint.",
|
||||
group="AI grading",
|
||||
),
|
||||
ConfigField(
|
||||
"news_grade_prompt",
|
||||
"Grading prompt specification",
|
||||
type="text",
|
||||
default=GRADE_PROMPT_SPEC,
|
||||
help=(
|
||||
"The exact rubric sent to the grading model at temperature 0. "
|
||||
"The cleaned Title, Description and Content are appended "
|
||||
"automatically. It must instruct the model to return only a "
|
||||
"single integer from 1 to 10."
|
||||
),
|
||||
group="AI grading",
|
||||
),
|
||||
ConfigField(
|
||||
"news_grade_threshold",
|
||||
"Grade threshold (1-10)",
|
||||
@@ -105,7 +367,11 @@ class NewsService(BaseService):
|
||||
default=GRADE_THRESHOLD_DEFAULT,
|
||||
minimum=1,
|
||||
maximum=10,
|
||||
help="Articles graded at or above this are auto-published; below go to draft.",
|
||||
help=(
|
||||
"Articles whose effective score (AI grade plus the unique-image "
|
||||
"bonus minus the thin-content penalty) reaches this are "
|
||||
"auto-published; below go to draft."
|
||||
),
|
||||
group="AI grading",
|
||||
),
|
||||
ConfigField(
|
||||
@@ -154,205 +420,365 @@ class NewsService(BaseService):
|
||||
updated_count = 0
|
||||
draft_count = 0
|
||||
failed_count = 0
|
||||
rejected_count = 0
|
||||
skipped_count = 0
|
||||
|
||||
for article in articles:
|
||||
external_id = article.get("guid", "")
|
||||
if not external_id:
|
||||
continue
|
||||
async with net_guard.guarded_async_client(
|
||||
timeout=IMG_FETCH_TIMEOUT
|
||||
) as image_client:
|
||||
candidates_by_article: dict[str, list[ImageCandidate]] = {}
|
||||
pending: list[tuple[dict, str, ImageCandidate | None]] = []
|
||||
|
||||
if external_id in synced_ids:
|
||||
skipped_count += 1
|
||||
continue
|
||||
for article in articles:
|
||||
external_id = article.get("guid", "")
|
||||
if not external_id:
|
||||
continue
|
||||
if external_id in synced_ids:
|
||||
skipped_count += 1
|
||||
continue
|
||||
|
||||
grade = await self._grade_article(article, ai_url, ai_model, client)
|
||||
now = datetime.now(timezone.utc).isoformat()
|
||||
article_uid, is_new = self._resolve_uid(news_table, external_id)
|
||||
link = article.get("link", "")
|
||||
candidates: list[ImageCandidate] = []
|
||||
if link:
|
||||
raw_images = await _get_article_images(link, client)
|
||||
for src in raw_images[:IMG_PER_ARTICLE]:
|
||||
candidates.append(
|
||||
await _fetch_image_candidate(src, image_client)
|
||||
)
|
||||
candidates_by_article[article_uid] = candidates
|
||||
pending.append((article, article_uid, None))
|
||||
|
||||
if grade is None:
|
||||
grade_val = 0
|
||||
auto_published = False
|
||||
failed_count += 1
|
||||
sync_status = "grading_failed"
|
||||
elif grade < threshold:
|
||||
grade_val = grade
|
||||
auto_published = False
|
||||
draft_count += 1
|
||||
sync_status = "graded"
|
||||
else:
|
||||
grade_val = grade
|
||||
auto_published = True
|
||||
sync_status = "graded"
|
||||
_flag_shared_placeholders(candidates_by_article)
|
||||
|
||||
is_published = "published" if auto_published else "draft"
|
||||
existing = news_table.find_one(external_id=external_id)
|
||||
|
||||
if existing:
|
||||
existing_slug = existing.get("slug", "")
|
||||
news_table.update(
|
||||
{
|
||||
"id": existing["id"],
|
||||
"grade": grade_val,
|
||||
"status": is_published,
|
||||
"title": article.get("title", ""),
|
||||
"slug": existing_slug
|
||||
or make_combined_slug(
|
||||
article.get("title", "") or "news", existing["uid"]
|
||||
),
|
||||
"description": strip_html(
|
||||
article.get("description", "") or ""
|
||||
)[:5000],
|
||||
"url": article.get("link", ""),
|
||||
"source_name": article.get("feed_name", ""),
|
||||
"content": strip_html(article.get("content", "") or "")[
|
||||
:10000
|
||||
],
|
||||
"author": article.get("author", ""),
|
||||
"article_published": article.get("published", ""),
|
||||
"synced_at": now,
|
||||
},
|
||||
["id"],
|
||||
for article, article_uid, _ in pending:
|
||||
candidates = candidates_by_article[article_uid]
|
||||
ai_grade = await self._grade_article(
|
||||
article, ai_url, ai_model, client
|
||||
)
|
||||
updated_count += 1
|
||||
images_table.delete(news_uid=existing["uid"])
|
||||
article_uid = existing["uid"]
|
||||
else:
|
||||
article_uid = generate_uid()
|
||||
article_slug = make_combined_slug(
|
||||
article.get("title", "") or "news", article_uid
|
||||
)
|
||||
news_table.insert(
|
||||
{
|
||||
"uid": article_uid,
|
||||
"slug": article_slug,
|
||||
"external_id": external_id,
|
||||
"title": article.get("title", ""),
|
||||
"description": strip_html(
|
||||
article.get("description", "") or ""
|
||||
)[:5000],
|
||||
"url": article.get("link", ""),
|
||||
"image_url": "",
|
||||
"source_name": article.get("feed_name", ""),
|
||||
"grade": grade_val,
|
||||
"status": is_published,
|
||||
"content": strip_html(article.get("content", "") or "")[
|
||||
:10000
|
||||
],
|
||||
"author": article.get("author", ""),
|
||||
"article_published": article.get("published", ""),
|
||||
"synced_at": now,
|
||||
"deleted_at": None,
|
||||
"deleted_by": None,
|
||||
}
|
||||
)
|
||||
new_count += 1
|
||||
title = article.get("title", "")
|
||||
audit.record_system(
|
||||
"news.service.ingest",
|
||||
actor_kind="service",
|
||||
target_type="news",
|
||||
target_uid=article_uid,
|
||||
target_label=title,
|
||||
metadata={"source": article.get("feed_name", ""), "grade": grade_val},
|
||||
summary=f"news article {title} ingested",
|
||||
links=[audit.target("news", article_uid, title)],
|
||||
)
|
||||
audit.record_system(
|
||||
"news.service.publish" if auto_published else "news.service.draft",
|
||||
actor_kind="service",
|
||||
target_type="news",
|
||||
target_uid=article_uid,
|
||||
target_label=title,
|
||||
metadata={"grade": grade_val, "threshold": threshold},
|
||||
summary=f"news article {title} {'auto-published' if auto_published else 'held as draft'}",
|
||||
links=[audit.target("news", article_uid, title)],
|
||||
result = self._grade_article_full(
|
||||
article, ai_grade, candidates
|
||||
)
|
||||
|
||||
existing_sync = sync_table.find_one(external_id=external_id)
|
||||
if existing_sync:
|
||||
sync_table.update(
|
||||
{
|
||||
"id": existing_sync["id"],
|
||||
"status": sync_status,
|
||||
"synced_at": now,
|
||||
},
|
||||
["id"],
|
||||
if result.ai_grade is None:
|
||||
failed_count += 1
|
||||
sync_status = "grading_failed"
|
||||
elif not result.valid:
|
||||
rejected_count += 1
|
||||
sync_status = f"rejected_quality:{result.reject_reason}"
|
||||
elif result.effective_score < threshold:
|
||||
draft_count += 1
|
||||
sync_status = "graded"
|
||||
else:
|
||||
sync_status = "graded"
|
||||
|
||||
published = (
|
||||
result.valid and result.effective_score >= threshold
|
||||
)
|
||||
else:
|
||||
sync_table.insert(
|
||||
{
|
||||
"uid": generate_uid(),
|
||||
"external_id": external_id,
|
||||
"status": sync_status,
|
||||
"synced_at": now,
|
||||
}
|
||||
featured = (
|
||||
published
|
||||
and result.has_unique_image
|
||||
and result.effective_score >= FEATURE_MIN_SCORE
|
||||
)
|
||||
|
||||
synced_ids.add(external_id)
|
||||
saved_new = self._store_article(
|
||||
news_table,
|
||||
images_table,
|
||||
article,
|
||||
article_uid,
|
||||
result,
|
||||
published,
|
||||
featured,
|
||||
threshold,
|
||||
candidates,
|
||||
)
|
||||
if saved_new:
|
||||
new_count += 1
|
||||
else:
|
||||
updated_count += 1
|
||||
|
||||
link = article.get("link", "")
|
||||
if link:
|
||||
fresh_images = await _get_article_images(link, client)
|
||||
for img in fresh_images:
|
||||
images_table.insert(
|
||||
{
|
||||
"uid": generate_uid(),
|
||||
"news_uid": article_uid,
|
||||
"url": img["url"],
|
||||
"alt_text": img.get("alt_text", ""),
|
||||
"deleted_at": None,
|
||||
"deleted_by": None,
|
||||
}
|
||||
)
|
||||
self._record_sync(sync_table, article["guid"], sync_status)
|
||||
synced_ids.add(article["guid"])
|
||||
|
||||
self._apply_landing_selection(news_table, threshold)
|
||||
|
||||
self.log(
|
||||
f"New {new_count}, updated {updated_count}, draft {draft_count}, "
|
||||
f"grading failed {failed_count}, skipped {skipped_count}"
|
||||
f"rejected {rejected_count}, grading failed {failed_count}, "
|
||||
f"skipped {skipped_count}"
|
||||
)
|
||||
|
||||
def _resolve_uid(self, news_table, external_id: str) -> tuple[str, bool]:
|
||||
existing = news_table.find_one(external_id=external_id)
|
||||
if existing:
|
||||
return existing["uid"], False
|
||||
return generate_uid(), True
|
||||
|
||||
def _grade_article_full(
|
||||
self,
|
||||
article: dict,
|
||||
ai_grade: int | None,
|
||||
candidates: list[ImageCandidate],
|
||||
) -> ArticleGrade:
|
||||
title = clean_news_text(article.get("title", "") or "")
|
||||
description = clean_news_text(article.get("description", "") or "")
|
||||
content = clean_news_text(article.get("content", "") or "")
|
||||
body = f"{description} {content}".strip()
|
||||
url = article.get("link", "") or ""
|
||||
|
||||
has_unique_image = any(not c.is_placeholder for c in candidates)
|
||||
image_url = _primary_image(candidates)
|
||||
|
||||
if ai_grade is None:
|
||||
return ArticleGrade(
|
||||
ai_grade=None,
|
||||
effective_score=0,
|
||||
has_unique_image=has_unique_image,
|
||||
image_url=image_url,
|
||||
valid=False,
|
||||
reject_reason="grading_failed",
|
||||
candidates=candidates,
|
||||
)
|
||||
|
||||
reason = reliability_reason(title, body, url)
|
||||
if reason:
|
||||
return ArticleGrade(
|
||||
ai_grade=ai_grade,
|
||||
effective_score=ai_grade,
|
||||
has_unique_image=has_unique_image,
|
||||
image_url=image_url,
|
||||
valid=False,
|
||||
reject_reason=reason,
|
||||
candidates=candidates,
|
||||
)
|
||||
|
||||
body_marginal = len(body) < (MIN_BODY_CHARS * 2)
|
||||
score = effective_score(ai_grade, has_unique_image, body_marginal)
|
||||
return ArticleGrade(
|
||||
ai_grade=ai_grade,
|
||||
effective_score=score,
|
||||
has_unique_image=has_unique_image,
|
||||
image_url=image_url,
|
||||
valid=True,
|
||||
reject_reason="",
|
||||
candidates=candidates,
|
||||
)
|
||||
|
||||
def _store_article(
|
||||
self,
|
||||
news_table,
|
||||
images_table,
|
||||
article: dict,
|
||||
article_uid: str,
|
||||
result: ArticleGrade,
|
||||
published: bool,
|
||||
featured: bool,
|
||||
threshold: int,
|
||||
candidates: list[ImageCandidate],
|
||||
) -> bool:
|
||||
now = datetime.now(timezone.utc).isoformat()
|
||||
external_id = article.get("guid", "")
|
||||
title = clean_news_text(article.get("title", "") or "")[:500] or "news"
|
||||
description = clean_news_text(article.get("description", "") or "")[:5000]
|
||||
content = clean_news_text(article.get("content", "") or "")[:10000]
|
||||
status = "published" if published else "draft"
|
||||
existing = news_table.find_one(external_id=external_id)
|
||||
|
||||
if existing:
|
||||
existing_slug = existing.get("slug", "")
|
||||
featured_locked = existing.get("featured_locked", 0)
|
||||
landing_locked = existing.get("landing_locked", 0)
|
||||
update_row = {
|
||||
"id": existing["id"],
|
||||
"grade": result.effective_score,
|
||||
"ai_grade": result.ai_grade or 0,
|
||||
"status": status,
|
||||
"title": title,
|
||||
"slug": existing_slug or make_combined_slug(title, existing["uid"]),
|
||||
"description": description,
|
||||
"url": article.get("link", ""),
|
||||
"source_name": article.get("feed_name", ""),
|
||||
"content": content,
|
||||
"author": article.get("author", ""),
|
||||
"article_published": article.get("published", ""),
|
||||
"image_url": result.image_url,
|
||||
"has_unique_image": 1 if result.has_unique_image else 0,
|
||||
"synced_at": now,
|
||||
}
|
||||
if not featured_locked:
|
||||
update_row["featured"] = 1 if featured else 0
|
||||
news_table.update(update_row, ["id"])
|
||||
images_table.delete(news_uid=existing["uid"])
|
||||
self._store_images(images_table, existing["uid"], candidates)
|
||||
return False
|
||||
|
||||
slug = make_combined_slug(title, article_uid)
|
||||
news_table.insert(
|
||||
{
|
||||
"uid": article_uid,
|
||||
"slug": slug,
|
||||
"external_id": external_id,
|
||||
"title": title,
|
||||
"description": description,
|
||||
"url": article.get("link", ""),
|
||||
"image_url": result.image_url,
|
||||
"has_unique_image": 1 if result.has_unique_image else 0,
|
||||
"source_name": article.get("feed_name", ""),
|
||||
"grade": result.effective_score,
|
||||
"ai_grade": result.ai_grade or 0,
|
||||
"status": status,
|
||||
"featured": 1 if featured else 0,
|
||||
"featured_locked": 0,
|
||||
"landing_locked": 0,
|
||||
"show_on_landing": 0,
|
||||
"content": content,
|
||||
"author": article.get("author", ""),
|
||||
"article_published": article.get("published", ""),
|
||||
"synced_at": now,
|
||||
"deleted_at": None,
|
||||
"deleted_by": None,
|
||||
}
|
||||
)
|
||||
self._store_images(images_table, article_uid, candidates)
|
||||
audit.record_system(
|
||||
"news.service.ingest",
|
||||
actor_kind="service",
|
||||
target_type="news",
|
||||
target_uid=article_uid,
|
||||
target_label=title,
|
||||
metadata={
|
||||
"source": article.get("feed_name", ""),
|
||||
"grade": result.effective_score,
|
||||
"ai_grade": result.ai_grade,
|
||||
"unique_image": result.has_unique_image,
|
||||
},
|
||||
summary=f"news article {title} ingested",
|
||||
links=[audit.target("news", article_uid, title)],
|
||||
)
|
||||
if result.valid:
|
||||
audit.record_system(
|
||||
"news.service.publish" if published else "news.service.draft",
|
||||
actor_kind="service",
|
||||
target_type="news",
|
||||
target_uid=article_uid,
|
||||
target_label=title,
|
||||
metadata={
|
||||
"grade": result.effective_score,
|
||||
"threshold": threshold,
|
||||
"featured": featured,
|
||||
},
|
||||
summary=(
|
||||
f"news article {title} "
|
||||
f"{'auto-published' if published else 'held as draft'}"
|
||||
),
|
||||
links=[audit.target("news", article_uid, title)],
|
||||
)
|
||||
else:
|
||||
audit.record_system(
|
||||
"news.service.reject",
|
||||
actor_kind="service",
|
||||
target_type="news",
|
||||
target_uid=article_uid,
|
||||
target_label=title,
|
||||
metadata={"reason": result.reject_reason},
|
||||
summary=f"news article {title} rejected ({result.reject_reason})",
|
||||
links=[audit.target("news", article_uid, title)],
|
||||
)
|
||||
return True
|
||||
|
||||
def _store_images(
|
||||
self, images_table, news_uid: str, candidates: list[ImageCandidate]
|
||||
) -> None:
|
||||
for candidate in candidates:
|
||||
images_table.insert(
|
||||
{
|
||||
"uid": generate_uid(),
|
||||
"news_uid": news_uid,
|
||||
"url": candidate.url,
|
||||
"alt_text": candidate.alt_text,
|
||||
"phash": candidate.phash,
|
||||
"width": candidate.width,
|
||||
"height": candidate.height,
|
||||
"is_placeholder": 1 if candidate.is_placeholder else 0,
|
||||
"deleted_at": None,
|
||||
"deleted_by": None,
|
||||
}
|
||||
)
|
||||
|
||||
def _record_sync(self, sync_table, external_id: str, status: str) -> None:
|
||||
now = datetime.now(timezone.utc).isoformat()
|
||||
existing = sync_table.find_one(external_id=external_id)
|
||||
if existing:
|
||||
sync_table.update(
|
||||
{"id": existing["id"], "status": status, "synced_at": now}, ["id"]
|
||||
)
|
||||
else:
|
||||
sync_table.insert(
|
||||
{
|
||||
"uid": generate_uid(),
|
||||
"external_id": external_id,
|
||||
"status": status,
|
||||
"synced_at": now,
|
||||
}
|
||||
)
|
||||
|
||||
def _apply_landing_selection(self, news_table, threshold: int) -> None:
|
||||
cutoff = (
|
||||
datetime.now(timezone.utc) - timedelta(days=LANDING_RECENCY_DAYS)
|
||||
).isoformat()
|
||||
managed = list(
|
||||
news_table.find(
|
||||
deleted_at=None,
|
||||
status="published",
|
||||
featured=1,
|
||||
has_unique_image=1,
|
||||
landing_locked=0,
|
||||
synced_at={">=": cutoff},
|
||||
order_by=["-grade", "-synced_at"],
|
||||
)
|
||||
)
|
||||
chosen: list[str] = []
|
||||
for article in managed:
|
||||
if (
|
||||
len(chosen) < LANDING_MAX
|
||||
and article.get("grade", 0) >= LANDING_MIN_SCORE
|
||||
):
|
||||
chosen.append(article["uid"])
|
||||
chosen_set = set(chosen)
|
||||
for article in managed:
|
||||
target = 1 if article["uid"] in chosen_set else 0
|
||||
if article.get("show_on_landing", 0) != target:
|
||||
news_table.update(
|
||||
{"uid": article["uid"], "show_on_landing": target}, ["uid"]
|
||||
)
|
||||
if target:
|
||||
audit.record_system(
|
||||
"news.service.landing",
|
||||
actor_kind="service",
|
||||
target_type="news",
|
||||
target_uid=article["uid"],
|
||||
target_label=article.get("title"),
|
||||
metadata={"grade": article.get("grade", 0)},
|
||||
summary=(
|
||||
f"news article {article.get('title')} promoted to landing"
|
||||
),
|
||||
links=[
|
||||
audit.target(
|
||||
"news", article["uid"], article.get("title")
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
async def _grade_article(
|
||||
self, article: dict, ai_url: str, ai_model: str, client: httpx.AsyncClient
|
||||
) -> int | None:
|
||||
title = (article.get("title", "") or "")[:500]
|
||||
description = strip_html(article.get("description", "") or "")[:1000]
|
||||
content = strip_html(article.get("content", "") or "")[:1500]
|
||||
title = clean_news_text(article.get("title", "") or "")[:500]
|
||||
description = clean_news_text(article.get("description", "") or "")[:1000]
|
||||
content = clean_news_text(article.get("content", "") or "")[:1500]
|
||||
|
||||
spec = self.get_config().get("news_grade_prompt", "") or GRADE_PROMPT_SPEC
|
||||
prompt = (
|
||||
"You are an expert content strategist and editor for DevPlace, a "
|
||||
"server-rendered social network for software developers. Decide how "
|
||||
"strongly a given article should be published, expressed as a single "
|
||||
"grade.\n\n"
|
||||
"Site context:\n"
|
||||
"- Audience: software developers, hobbyists and professionals, "
|
||||
"interested in software development, AI agents, open-source tools, "
|
||||
"systems programming, and self-hosting.\n"
|
||||
"- Mission: deliver deep technical insight, critical analysis, and "
|
||||
"practical value; build authority and reader trust; avoid low-effort, "
|
||||
"rehashed, or clickbait content.\n"
|
||||
"- Tone: technical, practical, nuanced, skeptical of hype; no "
|
||||
"sensationalism or unverified claims.\n\n"
|
||||
"Evaluate the article against these weighted criteria and weigh them "
|
||||
"internally:\n"
|
||||
"1. Relevance & alignment to the developer niche and audience needs "
|
||||
"(20%).\n"
|
||||
"2. Quality & accuracy: depth, originality, factual correctness, "
|
||||
"sourcing, technical correctness, clear writing (25%).\n"
|
||||
"3. Engagement & readability: strong title/intro/flow, useful "
|
||||
"examples, appropriate length (15%).\n"
|
||||
"4. Originality & uniqueness: new perspective, data, or analysis; not "
|
||||
"duplicated generic web content (15%).\n"
|
||||
"5. SEO & discoverability: keyword opportunity without stuffing, "
|
||||
"shareable (10%).\n"
|
||||
"6. Ethics, legality & risk: no misinformation, plagiarism, or "
|
||||
"brand-safety issues; controversial topics handled with nuance and "
|
||||
"evidence (10%).\n"
|
||||
"7. Strategic fit for the site (5%).\n\n"
|
||||
"Convert your overall judgment to a single integer grade from 1 to "
|
||||
"10:\n"
|
||||
"- 9-10: excellent, publish as-is.\n"
|
||||
"- 7-8: good, worth publishing.\n"
|
||||
"- 5-6: marginal, weak fit or quality.\n"
|
||||
"- 1-4: reject, low quality or off-topic.\n\n"
|
||||
"Return ONLY a single integer from 1 to 10, nothing else.\n\n"
|
||||
f"{spec}\n\n"
|
||||
f"Title: {title}\n"
|
||||
f"Description: {description}\n"
|
||||
f"Content: {content}"
|
||||
|
||||
Reference in New Issue
Block a user