# retoor <retoor@molodetz.nl>
import asyncio
import json
import logging
import shutil
import sqlite3
import sys
from datetime import datetime, timezone
from pathlib import Path
from devplacepy import config
from devplacepy.attachments import _directory_for
from devplacepy.services.backup import store
from devplacepy.services.devii.tasks.schedule import next_run as schedule_next_run
from devplacepy.services.devii.tasks.schedule import now_utc, to_iso
from devplacepy.services.jobs import queue
from devplacepy.services.jobs.base import JobService, _human_bytes
from devplacepy.utils import generate_uid
logger = logging.getLogger(__name__)
WORKER_MODULE = "devplacepy.services.jobs.backup_worker"
class BackupService(JobService):
kind = "backup"
title = "Backup"
description = (
"Builds compressed tar.gz backups of selected data targets in a subprocess off the "
"request path, snapshots the database consistently, records full statistics, fires "
"scheduled backups, and rotates them by retention."
)
def __init__(self):
super().__init__(name="backup", interval_seconds=15)
async def run_once(self) -> None:
await super().run_once()
try:
self._fire_due_schedules()
except Exception as exc:
self.log(f"Schedule pass failed: {exc}")
async def process(self, job: dict) -> dict:
from devplacepy.services.audit import record as audit
payload = job["payload"]
target = payload.get("target", "")
schedule_uid = payload.get("schedule_uid", "")
keep_last = int(payload.get("keep_last") or 0)
job_uid = job["uid"]
record = store.get_backup_by_job(job_uid)
if record is None:
backup_uid = store.create_backup(
target=target,
created_by=job.get("owner_id", ""),
job_uid=job_uid,
schedule_uid=schedule_uid,
)
record = store.get_backup(backup_uid)
backup_uid = record["uid"]
if not store.is_valid_target(target):
store.fail_backup(backup_uid, f"unknown backup target: {target}")
raise ValueError(f"unknown backup target: {target}")
store.mark_running(backup_uid)
staging = config.BACKUP_STAGING_DIR / job_uid
try:
sources = await asyncio.to_thread(self._materialize, target, staging)
spec_path = staging / "spec.json"
await asyncio.to_thread(
spec_path.write_text, json.dumps({"sources": sources})
)
final_dir = config.BACKUPS_DIR / _directory_for(backup_uid)
await asyncio.to_thread(final_dir.mkdir, parents=True, exist_ok=True)
filename = self._archive_name(target, backup_uid)
tmp_path = final_dir / f"{generate_uid()}.partial"
stats = await self._run_worker(spec_path, tmp_path)
final_path = final_dir / filename
await asyncio.to_thread(tmp_path.replace, final_path)
except Exception as exc:
store.fail_backup(backup_uid, str(exc) or exc.__class__.__name__)
audit.record_system(
"job.backup.failed",
actor_kind="system",
origin="scheduler" if schedule_uid else "web",
result="failure",
target_type="backup",
target_uid=backup_uid,
metadata={"target": target, "schedule_uid": schedule_uid},
summary=f"backup {target} failed",
links=[audit.job(job_uid)],
)
raise
finally:
await asyncio.to_thread(shutil.rmtree, staging, ignore_errors=True)
store.finalize_backup(
backup_uid,
filename=filename,
local_path=str(final_path),
stats=stats,
)
removed = store.rotate_schedule(schedule_uid, keep_last)
audit.record_system(
"job.backup.complete",
actor_kind="system",
origin="scheduler" if schedule_uid else "web",
target_type="backup",
target_uid=backup_uid,
metadata={
"target": target,
"schedule_uid": schedule_uid,
"bytes_out": stats["bytes_out"],
"sha256": stats["sha256"],
"file_count": stats["file_count"],
"rotated": removed,
},
summary=f"backup {target} completed ({_human_bytes(stats['bytes_out'])})",
links=[audit.job(job_uid)],
)
return {
"backup_uid": backup_uid,
"target": target,
"filename": filename,
"local_path": str(final_path),
"sha256": stats["sha256"],
"file_count": stats["file_count"],
"dir_count": stats["dir_count"],
"bytes_in": stats["bytes_in"],
"bytes_out": stats["bytes_out"],
"item_count": stats["file_count"],
}
def cleanup(self, job: dict) -> None:
shutil.rmtree(config.BACKUP_STAGING_DIR / job["uid"], ignore_errors=True)
def _materialize(self, target: str, staging: Path) -> list[dict]:
staging.mkdir(parents=True, exist_ok=True)
sources: list[dict] = []
if target in ("database", "full"):
snapshot = staging / "db_snapshot"
self._snapshot_databases(snapshot)
sources.append({"root": "database", "path": str(snapshot)})
if target in ("uploads", "full"):
sources.append({"root": "uploads", "path": str(config.UPLOADS_DIR)})
if target in ("keys", "full"):
sources.append({"root": "keys", "path": str(config.KEYS_DIR)})
return sources
def _snapshot_databases(self, dest: Path) -> None:
dest.mkdir(parents=True, exist_ok=True)
database_file = Path(str(config.DATABASE_URL).replace("sqlite:///", ""))
for source in (database_file, config.DEVII_TASKS_DB, config.DEVII_LESSONS_DB):
if Path(source).exists():
self._sqlite_backup(Path(source), dest / Path(source).name)
def _sqlite_backup(self, source: Path, destination: Path) -> None:
origin = sqlite3.connect(f"file:{source}?mode=ro", uri=True)
try:
target = sqlite3.connect(str(destination))
try:
origin.backup(target)
finally:
target.close()
finally:
origin.close()
async def _run_worker(self, spec_path: Path, output_path: Path) -> dict:
proc = await asyncio.create_subprocess_exec(
sys.executable,
"-m",
WORKER_MODULE,
str(spec_path),
str(output_path),
cwd=str(config.BASE_DIR),
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
)
out, err = await proc.communicate()
if proc.returncode != 0:
raise RuntimeError(
f"backup worker exited {proc.returncode}: "
f"{err.decode('utf-8', 'replace')[:500]}"
)
payload = out.decode("utf-8").strip()
if not payload:
raise RuntimeError(
f"backup worker produced no output: "
f"{err.decode('utf-8', 'replace')[:500]}"
)
return json.loads(payload)
def _archive_name(self, target: str, backup_uid: str) -> str:
stamp = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
tail = backup_uid.replace("-", "")[-8:]
return f"{target}-{stamp}-{tail}.tar.gz"
def _fire_due_schedules(self) -> None:
reference = to_iso(now_utc())
for schedule in store.list_due_schedules(reference):
self._enqueue_scheduled(schedule, reference)
def _enqueue_scheduled(self, schedule: dict, reference: str) -> None:
target = schedule["target"]
if not store.is_valid_target(target):
self.log(f"Schedule {schedule['uid']} skipped: unknown target {target}")
return
keep_last = int(schedule.get("keep_last") or 0)
created_by = schedule.get("created_by", "")
job_uid = queue.enqueue(
"backup",
{
"target": target,
"schedule_uid": schedule["uid"],
"created_by": created_by,
"keep_last": keep_last,
},
owner_kind="system",
owner_id=created_by,
preferred_name=f"{schedule.get('name', target)} ({target})",
)
store.create_backup(
target=target,
created_by=created_by,
job_uid=job_uid,
schedule_uid=schedule["uid"],
)
moment = schedule_next_run(
schedule["kind"],
int(schedule.get("every_seconds") or 0),
schedule.get("cron") or None,
now_utc(),
)
store.set_schedule_runtime(
schedule["uid"],
next_run_at=to_iso(moment) if moment else "",
last_run_at=reference,
last_job_uid=job_uid,
run_count=int(schedule.get("run_count") or 0) + 1,
)
self.log(f"Scheduled backup '{schedule.get('name')}' enqueued ({job_uid})")