Compare commits

..
Author SHA1 Message Date
Typosaurus 1eeb54598f ticket #104 attempt 1 2026-07-23 02:32:33 +00:00
Typosaurus a0d573375a ticket #104 attempt 1 2026-07-23 02:25:31 +00:00
6 changed files with 125 additions and 213 deletions
File diff suppressed because one or more lines are too long
-21
View File
@@ -30,26 +30,6 @@ def migrate_bug_tables_to_issue_tables() -> None:
logger.info("Dropped table %s after migration", source_name) logger.info("Dropped table %s after migration", source_name)
def _add_message_unique_index() -> None:
if "messages" not in db.tables:
return
with db:
db.query(
"""
DELETE FROM messages WHERE id NOT IN (
SELECT MIN(id) FROM messages GROUP BY sender_uid, receiver_uid, content
)
"""
)
_index(
db,
"messages",
"idx_messages_unique_sender_receiver_content",
["sender_uid", "receiver_uid", "content"],
unique=True,
)
def init_db(): def init_db():
tables = db.tables tables = db.tables
_index(db, "users", "idx_users_username", ["username"]) _index(db, "users", "idx_users_username", ["username"])
@@ -151,7 +131,6 @@ def init_db():
"idx_messages_conversation_rev", "idx_messages_conversation_rev",
["receiver_uid", "sender_uid"], ["receiver_uid", "sender_uid"],
) )
_add_message_unique_index()
_index(db, "notifications", "idx_notifications_user", ["user_uid"]) _index(db, "notifications", "idx_notifications_user", ["user_uid"])
_index(db, "notifications", "idx_notifications_user_read", ["user_uid", "read"]) _index(db, "notifications", "idx_notifications_user_read", ["user_uid", "read"])
_index(db, "push_registration", "idx_push_registration_user", ["user_uid"]) _index(db, "push_registration", "idx_push_registration_user", ["user_uid"])
+1 -1
View File
@@ -88,7 +88,7 @@ def gateway_complete(
{"role": "system", "content": system}, {"role": "system", "content": system},
{"role": "user", "content": text}, {"role": "user", "content": text},
], ],
"temperature": 0.0, "temperature": 0.1,
} }
headers = { headers = {
"Content-Type": "application/json", "Content-Type": "application/json",
+18 -15
View File
@@ -22,6 +22,8 @@ from .pdf import MAX_PDF_BYTES, extract_pdf_text, is_pdf
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
_pw_lock = asyncio.Lock()
RSEARCH_URL = "https://rsearch.app.molodetz.nl" RSEARCH_URL = "https://rsearch.app.molodetz.nl"
RSEARCH_TIMEOUT_SECONDS = 45.0 RSEARCH_TIMEOUT_SECONDS = 45.0
FETCH_TIMEOUT_SECONDS = 20.0 FETCH_TIMEOUT_SECONDS = 20.0
@@ -172,13 +174,7 @@ async def _render_with_playwright(
) -> tuple[str, str, int, list[tuple[str, str]]]: ) -> tuple[str, str, int, list[tuple[str, str]]]:
from playwright.async_api import async_playwright from playwright.async_api import async_playwright
own_browser = browser is None async def _render(browser) -> tuple[str, str, int, list[tuple[str, str]]]:
if own_browser:
pw = await async_playwright().__aenter__()
browser = await pw.chromium.launch(
headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"]
)
try:
context = await browser.new_context(user_agent=USER_AGENT) context = await browser.new_context(user_agent=USER_AGENT)
page = await context.new_page() page = await context.new_page()
response = await page.goto(url, wait_until="load", timeout=30000) response = await page.goto(url, wait_until="load", timeout=30000)
@@ -189,10 +185,18 @@ async def _render_with_playwright(
await context.close() await context.close()
extracted = extract_html(content, base_url=url) extracted = extract_html(content, base_url=url)
return extracted.title, extracted.text, status, extracted.links return extracted.title, extracted.text, status, extracted.links
if browser is None:
async with async_playwright() as pw:
browser = await pw.chromium.launch(
headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"]
)
try:
return await _render(browser)
finally: finally:
if own_browser:
await browser.close() await browser.close()
await pw.__aexit__(None, None, None) else:
return await _render(browser)
async def fetch_page(url: str, depth: int, browser=None) -> CrawledPage | None: async def fetch_page(url: str, depth: int, browser=None) -> CrawledPage | None:
@@ -242,8 +246,12 @@ async def fetch_page(url: str, depth: int, browser=None) -> CrawledPage | None:
except (LookupError, ValueError) as exc: except (LookupError, ValueError) as exc:
logger.info("deepsearch decode failed for %s: %s", url, exc) logger.info("deepsearch decode failed for %s: %s", url, exc)
if len(text) < MIN_PAGE_CHARS: if len(text) < MIN_PAGE_CHARS:
async with _pw_lock:
try: try:
r_title, r_text, r_status, r_links = await _render_with_playwright(url, browser) r_title, r_text, r_status, r_links = await _render_with_playwright(url, browser)
except Exception as exc:
logger.info("deepsearch render failed for %s: %s", url, exc)
r_title, r_text, r_status, r_links = "", "", 0, []
if len(r_text) > len(text): if len(r_text) > len(text):
title, text, status, source, links = ( title, text, status, source, links = (
r_title or title, r_title or title,
@@ -252,8 +260,6 @@ async def fetch_page(url: str, depth: int, browser=None) -> CrawledPage | None:
"playwright", "playwright",
r_links, r_links,
) )
except Exception as exc:
logger.info("deepsearch render failed for %s: %s", url, exc)
if len(text) < MIN_PAGE_CHARS: if len(text) < MIN_PAGE_CHARS:
return None return None
return CrawledPage( return CrawledPage(
@@ -296,10 +302,9 @@ async def crawl(
total = min(len(level_candidates), max_pages) total = min(len(level_candidates), max_pages)
cancelled = False cancelled = False
pw = None
browser = None browser = None
async with async_playwright() as pw:
try: try:
pw = await async_playwright().__aenter__()
browser = await pw.chromium.launch( browser = await pw.chromium.launch(
headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"] headless=True, args=["--no-sandbox", "--disable-dev-shm-usage"]
) )
@@ -392,6 +397,4 @@ async def crawl(
finally: finally:
if browser is not None: if browser is not None:
await browser.close() await browser.close()
if pw is not None:
await pw.__aexit__(None, None, None)
return outcome return outcome
-50
View File
@@ -1,9 +1,6 @@
# retoor <retoor@molodetz.nl> # retoor <retoor@molodetz.nl>
import hashlib
import logging import logging
import time
from collections import OrderedDict
from datetime import datetime, timezone from datetime import datetime, timezone
from typing import Any, Optional from typing import Any, Optional
@@ -18,16 +15,12 @@ from devplacepy.utils import (
track_action, track_action,
) )
from devplacepy.services.audit import record as audit from devplacepy.services.audit import record as audit
from sqlalchemy.exc import IntegrityError
from devplacepy.services.correction import schedule_correction from devplacepy.services.correction import schedule_correction
from devplacepy.services.ai_modifier import schedule_modification from devplacepy.services.ai_modifier import schedule_modification
logger = logging.getLogger("messaging.persist") logger = logging.getLogger("messaging.persist")
MAX_CONTENT_LENGTH = 2000 MAX_CONTENT_LENGTH = 2000
DEDUP_WINDOW_SECONDS = 3
_content_cache: dict[str, tuple[float, str]] = OrderedDict()
def _slim_attachment(attachment: dict[str, Any]) -> dict[str, Any]: def _slim_attachment(attachment: dict[str, Any]) -> dict[str, Any]:
@@ -88,32 +81,9 @@ def persist_message(
sender_uid = sender["uid"] sender_uid = sender["uid"]
sender_username = sender.get("username", "") sender_username = sender.get("username", "")
content_hash = hashlib.sha256(
f"{sender_uid}:{receiver_uid}:{content}".encode()
).hexdigest()[:16]
now = time.time()
last_seen, cached_uid = _content_cache.get(content_hash, (0.0, None))
if now - last_seen < DEDUP_WINDOW_SECONDS and cached_uid is not None:
logger.debug(
"Dedup hit for message hash %s (original uid %s)", content_hash, cached_uid
)
cached = get_table("messages").find_one(uid=cached_uid)
if cached:
return {
"uid": cached["uid"],
"sender_uid": cached["sender_uid"],
"receiver_uid": cached["receiver_uid"],
"content": cached["content"],
"read": cached.get("read", False),
"created_at": cached["created_at"],
}
messages_table = get_table("messages") messages_table = get_table("messages")
msg_uid = generate_uid() msg_uid = generate_uid()
created_at = datetime.now(timezone.utc).isoformat() created_at = datetime.now(timezone.utc).isoformat()
try:
messages_table.insert( messages_table.insert(
{ {
"uid": msg_uid, "uid": msg_uid,
@@ -124,24 +94,6 @@ def persist_message(
"created_at": created_at, "created_at": created_at,
} }
) )
except IntegrityError:
existing = messages_table.find_one(
sender_uid=sender_uid, receiver_uid=receiver_uid, content=content
)
if not existing:
raise
logger.debug(
"Dedup via unique constraint for message (uid %s)", existing["uid"]
)
_content_cache[content_hash] = (time.time(), existing["uid"])
return {
"uid": existing["uid"],
"sender_uid": existing["sender_uid"],
"receiver_uid": existing["receiver_uid"],
"content": existing["content"],
"read": existing.get("read", False),
"created_at": existing["created_at"],
}
link_attachments(attachment_uids, "message", msg_uid) link_attachments(attachment_uids, "message", msg_uid)
schedule_correction(sender, "messages", msg_uid, request) schedule_correction(sender, "messages", msg_uid, request)
@@ -162,8 +114,6 @@ def persist_message(
) )
track_action(sender_uid, "message") track_action(sender_uid, "message")
_content_cache[content_hash] = (time.time(), msg_uid)
logger.info( logger.info(
"Message %s sent from %s to %s via %s", "Message %s sent from %s to %s via %s",
msg_uid, msg_uid,
-20
View File
@@ -166,23 +166,3 @@ def test_send_attachment_only_empty_content_succeeds(seeded_db):
refresh_snapshot() refresh_snapshot()
row = get_table("messages").find_one(uid=msg["uid"]) row = get_table("messages").find_one(uid=msg["uid"])
assert row["content"] == "" assert row["content"] == ""
def test_duplicate_message_returns_same_uid(seeded_db):
s, _ = _member()
receiver = _db_user("bob_test")["uid"]
content = _unique("dupmsg")
first = s.post(
f"{BASE_URL}/messages/send",
headers=JSON_audit_log,
data={"content": content, "receiver_uid": receiver},
).json()["data"]
second = s.post(
f"{BASE_URL}/messages/send",
headers=JSON_audit_log,
data={"content": content, "receiver_uid": receiver},
).json()["data"]
assert first["uid"] == second["uid"], "duplicate messages should return the same uid"