Add the staff stats dashboard

Backend: GET /api/stats/ (staff only) gathers psutil CPU/memory counters,
nvidia-smi GPU stats, the cached disk numbers and the running/finished
download + match jobs. Root logging now also writes a rotating file
(backend/logs/j621.log) so the dashboard can tail it, and psutil joins the
requirements. The storage payload computation is shared with the existing
storage endpoint.

Frontend: a /stats route + Stats nav entry for staff, polling every 2 s —
per-core CPU bars, memory and swap, GPUs (utilization, VRAM, temperature),
disk with the media/temp breakdown, active jobs with progress bars,
recently finished jobs with summaries, and the log tail with level colours
and an auto-scroll toggle. Section 4 of the roadmap is complete.
This commit is contained in:
2026-09-17 21:36:33 -05:00
parent 0fcc4e518a
commit b024fc52d7
11 changed files with 808 additions and 62 deletions
+237
View File
@@ -0,0 +1,237 @@
"""Runtime statistics for the staff dashboard.
Everything here is read-only and cheap enough to poll every couple of
seconds: psutil counters, an optional nvidia-smi query, the cached storage
numbers and the tail of the application log file.
"""
import logging
import os
import subprocess
import time
from pathlib import Path
import psutil
from django.conf import settings
logger = logging.getLogger(__name__)
NVIDIA_QUERY = (
"name,utilization.gpu,memory.used,memory.total,temperature.gpu"
)
LOG_TAIL_LINES = 140
RECENT_LIMIT = 6
# Prime the non-blocking CPU counters so the first request already has a delta.
psutil.cpu_percent(interval=None)
psutil.cpu_percent(interval=None, percpu=True)
def _as_int(value):
try:
return int(float(str(value).strip()))
except (TypeError, ValueError):
return None
def system_stats():
"""CPU, memory and uptime counters (non-blocking)."""
memory = psutil.virtual_memory()
swap = psutil.swap_memory()
try:
load = [round(value, 2) for value in psutil.getloadavg()]
except (AttributeError, OSError):
load = []
return {
"cpu": {
"percent": psutil.cpu_percent(interval=None),
"per_cpu": psutil.cpu_percent(interval=None, percpu=True),
"cores": psutil.cpu_count(logical=True),
"physical_cores": psutil.cpu_count(logical=False),
"load": load,
},
"memory": {
"total": memory.total,
"used": memory.used,
"available": memory.available,
"percent": memory.percent,
"swap_total": swap.total,
"swap_used": swap.used,
"swap_percent": swap.percent,
},
"uptime_seconds": int(time.time() - psutil.boot_time()),
}
def gpu_stats():
"""NVIDIA GPUs via nvidia-smi, or an empty list when unavailable."""
try:
result = subprocess.run(
[
"nvidia-smi",
f"--query-gpu={NVIDIA_QUERY}",
"--format=csv,noheader,nounits",
],
capture_output=True,
text=True,
timeout=4,
check=False,
)
except (OSError, subprocess.SubprocessError) as exc:
logger.debug("nvidia-smi unavailable: %s", exc)
return []
if result.returncode != 0:
return []
gpus = []
for line in result.stdout.strip().splitlines():
parts = [part.strip() for part in line.split(",")]
if len(parts) != 5:
continue
name, utilization, memory_used, memory_total, temperature = parts
used_mib = _as_int(memory_used)
total_mib = _as_int(memory_total)
gpus.append(
{
"name": name,
"utilization": _as_int(utilization),
"memory_used": used_mib * 1024 * 1024
if used_mib is not None
else None,
"memory_total": total_mib * 1024 * 1024
if total_mib is not None
else None,
"temperature": _as_int(temperature),
}
)
return gpus
def _download_job(task):
label = (
f"Post #{task.post_id}"
if task.post_id
else (task.filename or "Download")
)
detail = f"{task.downloaded} / {task.total}" if task.total else task.filename
return {
"kind": "download",
"id": str(task.id),
"label": label,
"status": task.status,
"progress": task.progress,
"downloaded": task.downloaded,
"total": task.total,
"detail": detail,
"created_at": task.created_at,
"updated_at": task.updated_at,
}
def _match_job(task):
progress = (
int(task.processed * 100 / task.total) if task.total else 0
)
return {
"kind": "match",
"id": str(task.id),
"label": f"e621 match scan ({task.scope})",
"status": task.status,
"progress": progress,
"processed": task.processed,
"total": task.total,
"detail": (
f"{task.processed}/{task.total} · {task.matched} matched · "
f"{task.not_found} not found · {task.deleted} deleted"
),
"created_at": task.created_at,
"updated_at": task.updated_at,
}
def jobs_stats():
"""Running/pending jobs plus the most recently finished ones."""
from apps.library.models import DownloadTask, MatchTask
active_statuses_download = [
DownloadTask.STATUS_PENDING,
DownloadTask.STATUS_DOWNLOADING,
]
active_statuses_match = [
MatchTask.STATUS_PENDING,
MatchTask.STATUS_RUNNING,
]
finished_download = [
DownloadTask.STATUS_COMPLETE,
DownloadTask.STATUS_ERROR,
DownloadTask.STATUS_CANCELLED,
]
finished_match = [
MatchTask.STATUS_COMPLETE,
MatchTask.STATUS_ERROR,
MatchTask.STATUS_CANCELLED,
]
active = [
_download_job(task)
for task in DownloadTask.objects.filter(
status__in=active_statuses_download
).order_by("created_at")[:RECENT_LIMIT]
]
active += [
_match_job(task)
for task in MatchTask.objects.filter(
status__in=active_statuses_match
).order_by("created_at")[:RECENT_LIMIT]
]
recent = []
for task in DownloadTask.objects.filter(
status__in=finished_download
).order_by("-updated_at")[:RECENT_LIMIT]:
entry = _download_job(task)
entry["summary"] = (
f"J-{task.library_item_id}"
if task.library_item_id
else (task.error[:160] if task.error else task.status)
)
recent.append(entry)
for task in MatchTask.objects.filter(
status__in=finished_match
).order_by("-updated_at")[:RECENT_LIMIT]:
entry = _match_job(task)
entry["summary"] = task.error[:160] if task.error else entry["detail"]
recent.append(entry)
recent.sort(key=lambda entry: entry["updated_at"], reverse=True)
return {
"active": active,
"recent": recent[:RECENT_LIMIT],
"active_count": len(active),
}
def tail_lines(path, count=LOG_TAIL_LINES):
"""Last `count` lines of a file without reading all of it."""
with open(path, "rb") as handle:
handle.seek(0, os.SEEK_END)
size = handle.tell()
block = 16384
data = b""
while size > 0 and data.count(b"\n") <= count:
step = min(block, size)
size -= step
handle.seek(size)
data = handle.read(step) + data
return data.decode("utf-8", errors="replace").splitlines()[-count:]
def log_tail():
path = Path(getattr(settings, "LOG_FILE", ""))
if not path or not path.exists():
return {"path": str(path), "lines": [], "missing": True}
try:
lines = tail_lines(path)
except OSError as exc:
logger.warning("Could not read the log file: %s", exc)
return {"path": str(path), "lines": [], "missing": True}
return {"path": str(path), "lines": lines, "missing": False}
+2 -1
View File
@@ -1,7 +1,8 @@
from django.urls import path
from .views import StatusView
from .views import StatsView, StatusView
urlpatterns = [
path("status/", StatusView.as_view(), name="status"),
path("stats/", StatsView.as_view(), name="stats"),
]
+27 -1
View File
@@ -1,10 +1,12 @@
import time
from django.conf import settings
from rest_framework.permissions import AllowAny
from rest_framework.exceptions import PermissionDenied
from rest_framework.permissions import AllowAny, IsAuthenticated
from rest_framework.response import Response
from rest_framework.views import APIView
from . import stats
from .system_info import get_os_info
@@ -40,3 +42,27 @@ class StatusView(APIView):
}
payload["server_time_ms"] = round((time.perf_counter() - started) * 1000, 1)
return Response(payload)
class StatsView(APIView):
"""Staff-only runtime statistics for the dashboard."""
permission_classes = [IsAuthenticated]
def get(self, request):
user = request.user
if not (
user.is_staff or user.is_superuser or user.role == user.ROLE_STAFF
):
raise PermissionDenied("Staff only.")
from apps.library.tools import storage_info
return Response(
{
"system": stats.system_stats(),
"gpus": stats.gpu_stats(),
"disk": storage_info(),
"jobs": stats.jobs_stats(),
"log": stats.log_tail(),
}
)