# retoor <retoor@molodetz.nl>
import shutil
import time
from datetime import datetime, timezone
from pathlib import Path
from devplacepy import config
from devplacepy.database import _index, db, get_table
from devplacepy.utils import generate_uid
BACKUP_TARGETS: dict[str, dict] = {
"database": {
"label": "Database",
"description": "Consistent snapshot of the SQLite database and the Devii task and lesson databases.",
},
"uploads": {
"label": "Uploads",
"description": "All attachments and project files under the uploads directory.",
},
"keys": {
"label": "Keys and config",
"description": "VAPID notification keys and other small config artifacts.",
},
"full": {
"label": "Full data directory",
"description": "Database, uploads, and keys in one archive, excluding regenerable staging, locks, and caches.",
},
}
STATUS_PENDING = "pending"
STATUS_RUNNING = "running"
STATUS_DONE = "done"
STATUS_FAILED = "failed"
STORAGE_CACHE_TTL_SECONDS = 30
_storage_cache: dict = {"at": 0.0, "data": None}
def now_iso() -> str:
return datetime.now(timezone.utc).isoformat()
def is_valid_target(key: str) -> bool:
return key in BACKUP_TARGETS
def target_label(key: str) -> str:
return BACKUP_TARGETS.get(key, {}).get("label", key)
def human_bytes(size: int) -> str:
value = float(max(0, int(size or 0)))
for unit in ("B", "KB", "MB", "GB", "TB"):
if value < 1024 or unit == "TB":
return f"{value:.0f} {unit}" if unit == "B" else f"{value:.1f} {unit}"
value /= 1024
return f"{value:.1f} TB"
def ensure_tables() -> None:
backups = get_table("backups")
for column, example in (
("uid", ""),
("job_uid", ""),
("target", ""),
("label", ""),
("status", STATUS_PENDING),
("filename", ""),
("local_path", ""),
("size_bytes", 0),
("bytes_in", 0),
("file_count", 0),
("dir_count", 0),
("sha256", ""),
("schedule_uid", ""),
("created_by", ""),
("created_at", ""),
("completed_at", ""),
("error", ""),
):
if not backups.has_column(column):
backups.create_column_by_example(column, example)
schedules = get_table("backup_schedules")
for column, example in (
("uid", ""),
("name", ""),
("target", ""),
("kind", "interval"),
("every_seconds", 0),
("cron", ""),
("enabled", 1),
("keep_last", 0),
("next_run_at", ""),
("last_run_at", ""),
("last_job_uid", ""),
("run_count", 0),
("created_by", ""),
("created_at", ""),
("updated_at", ""),
("deleted_at", None),
("deleted_by", None),
):
if not schedules.has_column(column):
schedules.create_column_by_example(column, example)
_index(db, "backups", "idx_backups_status", ["status"])
_index(db, "backups", "idx_backups_job", ["job_uid"])
_index(db, "backups", "idx_backups_schedule", ["schedule_uid", "created_at"])
_index(db, "backup_schedules", "idx_backup_schedules_enabled", ["enabled"])
def create_backup(
*, target: str, created_by: str, job_uid: str, schedule_uid: str = ""
) -> str:
uid = generate_uid()
get_table("backups").insert(
{
"uid": uid,
"job_uid": job_uid,
"target": target,
"label": target_label(target),
"status": STATUS_PENDING,
"filename": "",
"local_path": "",
"size_bytes": 0,
"bytes_in": 0,
"file_count": 0,
"dir_count": 0,
"sha256": "",
"schedule_uid": schedule_uid,
"created_by": created_by,
"created_at": now_iso(),
"completed_at": "",
"error": "",
}
)
return uid
def get_backup(uid: str) -> dict | None:
if "backups" not in db.tables:
return None
return get_table("backups").find_one(uid=uid)
def get_backup_by_job(job_uid: str) -> dict | None:
if "backups" not in db.tables:
return None
return get_table("backups").find_one(job_uid=job_uid)
def list_backups(limit: int = 100) -> list[dict]:
if "backups" not in db.tables:
return []
return list(get_table("backups").find(order_by=["-created_at"], _limit=limit))
def mark_running(uid: str) -> None:
get_table("backups").update(
{"uid": uid, "status": STATUS_RUNNING}, ["uid"]
)
def finalize_backup(uid: str, *, filename: str, local_path: str, stats: dict) -> None:
get_table("backups").update(
{
"uid": uid,
"status": STATUS_DONE,
"filename": filename,
"local_path": local_path,
"size_bytes": int(stats.get("bytes_out", 0)),
"bytes_in": int(stats.get("bytes_in", 0)),
"file_count": int(stats.get("file_count", 0)),
"dir_count": int(stats.get("dir_count", 0)),
"sha256": stats.get("sha256", ""),
"completed_at": now_iso(),
"error": "",
},
["uid"],
)
def fail_backup(uid: str, error: str) -> None:
get_table("backups").update(
{
"uid": uid,
"status": STATUS_FAILED,
"completed_at": now_iso(),
"error": error[:2000],
},
["uid"],
)
def delete_backup(uid: str) -> dict | None:
row = get_backup(uid)
if not row:
return None
_unlink_archive(row)
get_table("backups").delete(uid=uid)
return row
def rotate_schedule(schedule_uid: str, keep_last: int) -> int:
if keep_last <= 0 or not schedule_uid:
return 0
rows = list(
get_table("backups").find(
schedule_uid=schedule_uid,
status=STATUS_DONE,
order_by=["-created_at"],
)
)
removed = 0
for row in rows[keep_last:]:
delete_backup(row["uid"])
removed += 1
return removed
def prune_orphans() -> int:
removed = 0
for row in list_backups(limit=100000):
if row.get("status") != STATUS_DONE:
continue
local_path = row.get("local_path") or ""
if not local_path or not Path(local_path).is_file():
get_table("backups").delete(uid=row["uid"])
removed += 1
return removed
def clear_all() -> int:
rows = list_backups(limit=100000)
for row in rows:
_unlink_archive(row)
get_table("backups").delete()
return len(rows)
def _unlink_archive(row: dict) -> None:
local_path = row.get("local_path") or ""
if local_path:
Path(local_path).unlink(missing_ok=True)
def create_schedule(
*,
name: str,
target: str,
kind: str,
every_seconds: int,
cron: str,
keep_last: int,
created_by: str,
next_run_at: str,
) -> str:
uid = generate_uid()
get_table("backup_schedules").insert(
{
"uid": uid,
"name": name,
"target": target,
"kind": kind,
"every_seconds": every_seconds,
"cron": cron,
"enabled": 1,
"keep_last": keep_last,
"next_run_at": next_run_at,
"last_run_at": "",
"last_job_uid": "",
"run_count": 0,
"created_by": created_by,
"created_at": now_iso(),
"updated_at": now_iso(),
"deleted_at": None,
"deleted_by": None,
}
)
return uid
def get_schedule(uid: str) -> dict | None:
if "backup_schedules" not in db.tables:
return None
return get_table("backup_schedules").find_one(uid=uid, deleted_at=None)
def list_schedules() -> list[dict]:
if "backup_schedules" not in db.tables:
return []
return list(
get_table("backup_schedules").find(
deleted_at=None, order_by=["-created_at"]
)
)
def list_due_schedules(reference: str) -> list[dict]:
if "backup_schedules" not in db.tables:
return []
rows = get_table("backup_schedules").find(deleted_at=None, enabled=1)
return [
row
for row in rows
if (row.get("next_run_at") or "") and row["next_run_at"] <= reference
]
def update_schedule(uid: str, changes: dict) -> None:
changes = {**changes, "uid": uid, "updated_at": now_iso()}
get_table("backup_schedules").update(changes, ["uid"])
def set_schedule_runtime(
uid: str, *, next_run_at: str, last_run_at: str, last_job_uid: str, run_count: int
) -> None:
get_table("backup_schedules").update(
{
"uid": uid,
"next_run_at": next_run_at,
"last_run_at": last_run_at,
"last_job_uid": last_job_uid,
"run_count": run_count,
"updated_at": now_iso(),
},
["uid"],
)
def delete_schedule(uid: str, deleted_by: str) -> bool:
row = get_schedule(uid)
if not row:
return False
get_table("backup_schedules").update(
{"uid": uid, "deleted_at": now_iso(), "deleted_by": deleted_by}, ["uid"]
)
return True
def _path_size(path: Path) -> tuple[int, int]:
if not path.exists():
return 0, 0
if path.is_file():
return path.stat().st_size, 1
total = 0
files = 0
for entry in path.rglob("*"):
try:
if entry.is_file() and not entry.is_symlink():
total += entry.stat().st_size
files += 1
except OSError:
continue
return total, files
def _storage_paths() -> list[tuple[str, str, Path]]:
database_file = Path(str(config.DATABASE_URL).replace("sqlite:///", ""))
return [
("database", "Database file", database_file),
("devii_tasks", "Devii tasks DB", config.DEVII_TASKS_DB),
("devii_lessons", "Devii lessons DB", config.DEVII_LESSONS_DB),
("uploads", "Uploads", config.UPLOADS_DIR),
("attachments", "Attachments", config.ATTACHMENTS_DIR),
("project_files", "Project files", config.PROJECT_FILES_DIR),
("keys", "Keys", config.KEYS_DIR),
("zips", "Zip archives", config.ZIPS_DIR),
("deepsearch", "DeepSearch", config.DEEPSEARCH_DIR),
("container_workspaces", "Container workspaces", config.CONTAINER_WORKSPACES_DIR),
("backups", "Backups", config.BACKUPS_DIR),
]
def compute_storage_stats() -> dict:
now = time.monotonic()
if (
_storage_cache["data"] is not None
and (now - _storage_cache["at"]) < STORAGE_CACHE_TTL_SECONDS
):
return _storage_cache["data"]
paths = []
for key, label, path in _storage_paths():
size, files = _path_size(path)
paths.append(
{
"key": key,
"label": label,
"path": str(path),
"size_bytes": size,
"size_human": human_bytes(size),
"file_count": files,
"exists": path.exists(),
}
)
data_size, data_files = _path_size(config.DATA_DIR)
backups_size, backups_files = _path_size(config.BACKUPS_DIR)
backup_count = len(
[b for b in list_backups(limit=100000) if b.get("status") == STATUS_DONE]
)
usage = shutil.disk_usage(str(config.DATA_DIR))
data = {
"paths": paths,
"data_dir": {
"path": str(config.DATA_DIR),
"size_bytes": data_size,
"size_human": human_bytes(data_size),
"file_count": data_files,
},
"backups_total": {
"count": backup_count,
"size_bytes": backups_size,
"size_human": human_bytes(backups_size),
"file_count": backups_files,
},
"disk": {
"total_bytes": usage.total,
"used_bytes": usage.used,
"free_bytes": usage.free,
"total_human": human_bytes(usage.total),
"used_human": human_bytes(usage.used),
"free_human": human_bytes(usage.free),
"used_percent": round(usage.used / usage.total * 100, 1)
if usage.total
else 0.0,
},
"generated_at": now_iso(),
}
_storage_cache["data"] = data
_storage_cache["at"] = now
return data