feat: add TTLCache, post/news count helpers, and avatar ETag caching across modules

This commit is contained in:
2026-05-23 04:20:27 +00:00
parent 12a3d3a1a7
commit 8ff4b351da
17 changed files with 298 additions and 242 deletions
+122 -133
View File
@@ -5,7 +5,7 @@ from datetime import datetime, timezone
import httpx
from devplacepy.database import get_table
from devplacepy.database import get_table, get_setting
from devplacepy.services.base import BaseService
from devplacepy.utils import generate_uid, make_combined_slug, strip_html
@@ -17,22 +17,13 @@ AI_MODEL_DEFAULT = "molodetz"
GRADE_THRESHOLD_DEFAULT = 7
def _get_setting(key: str, default: str) -> str:
table = get_table("site_settings")
row = table.find_one(key=key)
if row is None:
return default
return row.get("value", default)
def _get_ai_key() -> str:
key = os.environ.get("NEWS_AI_KEY")
if key:
return key
table = get_table("site_settings")
row = table.find_one(key="news_ai_key")
if row:
return row.get("value", "")
key = get_setting("news_ai_key", "")
if key:
return key
key = os.environ.get("OPENROUTER_API_KEY")
if key:
return key
@@ -51,12 +42,11 @@ def _extract_grade(text: str) -> int | None:
return None
async def _get_article_images(url: str) -> list[dict]:
async def _get_article_images(url: str, client: httpx.AsyncClient) -> list[dict]:
try:
async with httpx.AsyncClient(timeout=10.0) as client:
resp = await client.get(url)
resp.raise_for_status()
html = resp.text
resp = await client.get(url, timeout=10.0)
resp.raise_for_status()
html = resp.text
pattern = re.compile(r'<img[^>]+src=["\']([^"\']+)["\']', re.IGNORECASE)
matches = pattern.findall(html)
images = []
@@ -77,10 +67,10 @@ class NewsService(BaseService):
super().__init__(name="news", interval_seconds=3600)
async def run_once(self) -> None:
api_url = _get_setting("news_api_url", NEWS_API_URL_DEFAULT)
ai_url = _get_setting("news_ai_url", AI_URL_DEFAULT)
ai_model = _get_setting("news_ai_model", AI_MODEL_DEFAULT)
threshold = int(_get_setting("news_grade_threshold", str(GRADE_THRESHOLD_DEFAULT)))
api_url = get_setting("news_api_url", NEWS_API_URL_DEFAULT)
ai_url = get_setting("news_ai_url", AI_URL_DEFAULT)
ai_model = get_setting("news_ai_model", AI_MODEL_DEFAULT)
threshold = int(get_setting("news_grade_threshold", str(GRADE_THRESHOLD_DEFAULT)))
self.log(f"Fetching news from {api_url}")
async with httpx.AsyncClient(timeout=30.0) as client:
@@ -92,125 +82,125 @@ class NewsService(BaseService):
self.log(f"Failed to fetch news API: {e}")
return
articles = data.get("articles", [])
self.log(f"Received {len(articles)} articles")
articles = data.get("articles", [])
self.log(f"Received {len(articles)} articles")
news_table = get_table("news")
images_table = get_table("news_images")
sync_table = get_table("news_sync")
news_table = get_table("news")
images_table = get_table("news_images")
sync_table = get_table("news_sync")
synced_ids = set()
for entry in sync_table.find():
synced_ids.add(entry["external_id"])
synced_ids = set()
for entry in sync_table.find():
synced_ids.add(entry["external_id"])
new_count = 0
updated_count = 0
draft_count = 0
failed_count = 0
skipped_count = 0
new_count = 0
updated_count = 0
draft_count = 0
failed_count = 0
skipped_count = 0
for article in articles:
external_id = article.get("guid", "")
if not external_id:
continue
for article in articles:
external_id = article.get("guid", "")
if not external_id:
continue
if external_id in synced_ids:
skipped_count += 1
continue
if external_id in synced_ids:
skipped_count += 1
continue
grade = await self._grade_article(article, ai_url, ai_model)
now = datetime.now(timezone.utc).isoformat()
grade = await self._grade_article(article, ai_url, ai_model, client)
now = datetime.now(timezone.utc).isoformat()
if grade is None:
grade_val = 0
auto_published = False
failed_count += 1
sync_status = "grading_failed"
elif grade < threshold:
grade_val = grade
auto_published = False
draft_count += 1
sync_status = "graded"
else:
grade_val = grade
auto_published = True
sync_status = "graded"
if grade is None:
grade_val = 0
auto_published = False
failed_count += 1
sync_status = "grading_failed"
elif grade < threshold:
grade_val = grade
auto_published = False
draft_count += 1
sync_status = "graded"
else:
grade_val = grade
auto_published = True
sync_status = "graded"
is_published = "published" if auto_published else "draft"
existing = news_table.find_one(external_id=external_id)
is_published = "published" if auto_published else "draft"
existing = news_table.find_one(external_id=external_id)
if existing:
existing_slug = existing.get("slug", "")
news_table.update({
"id": existing["id"],
"grade": grade_val,
"status": is_published,
"title": article.get("title", ""),
"slug": existing_slug or make_combined_slug(article.get("title", "") or "news", existing["uid"]),
"description": strip_html(article.get("description", "") or "")[:5000],
"url": article.get("link", ""),
"source_name": article.get("feed_name", ""),
"content": strip_html(article.get("content", "") or "")[:10000],
"author": article.get("author", ""),
"article_published": article.get("published", ""),
"synced_at": now,
}, ["id"])
updated_count += 1
images_table.delete(news_uid=existing["uid"])
article_uid = existing["uid"]
else:
article_uid = generate_uid()
article_slug = make_combined_slug(article.get("title", "") or "news", article_uid)
news_table.insert({
"uid": article_uid,
"slug": article_slug,
"external_id": external_id,
"title": article.get("title", ""),
"description": strip_html(article.get("description", "") or "")[:5000],
"url": article.get("link", ""),
"image_url": "",
"source_name": article.get("feed_name", ""),
"grade": grade_val,
"status": is_published,
"content": strip_html(article.get("content", "") or "")[:10000],
"author": article.get("author", ""),
"article_published": article.get("published", ""),
"synced_at": now,
})
new_count += 1
existing_sync = sync_table.find_one(external_id=external_id)
if existing_sync:
sync_table.update({
"id": existing_sync["id"],
"status": sync_status,
"synced_at": now,
}, ["id"])
else:
sync_table.insert({
"uid": generate_uid(),
"external_id": external_id,
"status": sync_status,
"synced_at": now,
})
synced_ids.add(external_id)
link = article.get("link", "")
if link:
fresh_images = await _get_article_images(link)
for img in fresh_images:
images_table.insert({
"uid": generate_uid(),
"news_uid": article_uid,
"url": img["url"],
"alt_text": img.get("alt_text", ""),
if existing:
existing_slug = existing.get("slug", "")
news_table.update({
"id": existing["id"],
"grade": grade_val,
"status": is_published,
"title": article.get("title", ""),
"slug": existing_slug or make_combined_slug(article.get("title", "") or "news", existing["uid"]),
"description": strip_html(article.get("description", "") or "")[:5000],
"url": article.get("link", ""),
"source_name": article.get("feed_name", ""),
"content": strip_html(article.get("content", "") or "")[:10000],
"author": article.get("author", ""),
"article_published": article.get("published", ""),
"synced_at": now,
}, ["id"])
updated_count += 1
images_table.delete(news_uid=existing["uid"])
article_uid = existing["uid"]
else:
article_uid = generate_uid()
article_slug = make_combined_slug(article.get("title", "") or "news", article_uid)
news_table.insert({
"uid": article_uid,
"slug": article_slug,
"external_id": external_id,
"title": article.get("title", ""),
"description": strip_html(article.get("description", "") or "")[:5000],
"url": article.get("link", ""),
"image_url": "",
"source_name": article.get("feed_name", ""),
"grade": grade_val,
"status": is_published,
"content": strip_html(article.get("content", "") or "")[:10000],
"author": article.get("author", ""),
"article_published": article.get("published", ""),
"synced_at": now,
})
new_count += 1
existing_sync = sync_table.find_one(external_id=external_id)
if existing_sync:
sync_table.update({
"id": existing_sync["id"],
"status": sync_status,
"synced_at": now,
}, ["id"])
else:
sync_table.insert({
"uid": generate_uid(),
"external_id": external_id,
"status": sync_status,
"synced_at": now,
})
synced_ids.add(external_id)
link = article.get("link", "")
if link:
fresh_images = await _get_article_images(link, client)
for img in fresh_images:
images_table.insert({
"uid": generate_uid(),
"news_uid": article_uid,
"url": img["url"],
"alt_text": img.get("alt_text", ""),
})
self.log(f"New {new_count}, updated {updated_count}, draft {draft_count}, "
f"grading failed {failed_count}, skipped {skipped_count}")
async def _grade_article(self, article: dict, ai_url: str, ai_model: str) -> int | None:
async def _grade_article(self, article: dict, ai_url: str, ai_model: str, client: httpx.AsyncClient) -> int | None:
title = (article.get("title", "") or "")[:500]
description = strip_html(article.get("description", "") or "")[:1000]
content = strip_html(article.get("content", "") or "")[:1500]
@@ -238,12 +228,11 @@ class NewsService(BaseService):
headers["Authorization"] = f"Bearer {ai_key}"
try:
async with httpx.AsyncClient(timeout=15.0) as client:
resp = await client.post(ai_url, json=payload, headers=headers)
if resp.status_code != 200:
self.log(f"AI grading returned {resp.status_code}: {resp.text[:200]}")
resp.raise_for_status()
result = resp.json()
resp = await client.post(ai_url, json=payload, headers=headers, timeout=15.0)
if resp.status_code != 200:
self.log(f"AI grading returned {resp.status_code}: {resp.text[:200]}")
resp.raise_for_status()
result = resp.json()
text = result.get("choices", [{}])[0].get("message", {}).get("content", "")
if not text:
self.log(f"AI grading returned empty content for: {title[:60]}")