Duplicates, delete & storage, users page with J-ID avatars
Backend: - Perceptual hashes (aHash/dHash/pHash/wHash via imagehash, no imgdd) stored on items, computed on upload/download and by the new compute_visual_hashes command - Duplicates API: exact duplicates (multi-location items), visual matches for one item, union-find similarity groups with pagination - Delete API with ownership/staff checks, per-item and per-copy deletion, watched-folder path validation; storage overview and temp cleanup; file list accepts j_ids batches - Staged uploads are flagged visual_match with their library matches (threshold via VISUAL_MATCH_THRESHOLD) - Staff users API: list with upload counts, set role and avatar by J-ID; User.avatar FK with signed avatar URLs - Download threads close their DB connection and stale tasks are reaped, keeping behaviour Gunicorn-friendly Frontend: - /duplicates: exact duplicate groups with per-copy delete, visual similarity controls, search similar to a J-ID, paginated groups with selection, bulk delete and dismiss - /delete: storage cards, delete by J-ID with preview grid, temp cleanup - /users: staff directory with role selects and avatar J-ID inputs - Nav + command palette entries; top-bar avatar; upload cards and the metadata modal show library visual matches
This commit is contained in:
@@ -3,10 +3,12 @@
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from datetime import timedelta
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from django.conf import settings
|
||||
from django.db import connection
|
||||
from django.utils import timezone
|
||||
|
||||
from . import services
|
||||
@@ -14,6 +16,27 @@ from .models import DownloadTask
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
STALE_AFTER = timedelta(minutes=30)
|
||||
|
||||
|
||||
def reap_stale_downloads():
|
||||
"""Mark tasks left hanging by a recycled worker as failed.
|
||||
|
||||
Gunicorn recycles workers (--max-requests, timeouts); a download thread
|
||||
dies with its worker, so long-stuck tasks are surfaced as errors instead
|
||||
of pretending to run forever.
|
||||
"""
|
||||
cutoff = timezone.now() - STALE_AFTER
|
||||
return DownloadTask.objects.filter(
|
||||
status__in=[DownloadTask.STATUS_PENDING, DownloadTask.STATUS_DOWNLOADING],
|
||||
updated_at__lt=cutoff,
|
||||
).update(
|
||||
status=DownloadTask.STATUS_ERROR,
|
||||
error="The worker restarted before this download finished.",
|
||||
speed=None,
|
||||
updated_at=timezone.now(),
|
||||
)
|
||||
|
||||
|
||||
def start_download_task(task_id):
|
||||
thread = threading.Thread(target=run_download_task, args=(task_id,), daemon=True)
|
||||
@@ -79,6 +102,7 @@ def run_download_task(task_id):
|
||||
)
|
||||
item, _, location, _ = services.index_file(destination, folder)
|
||||
services.rename_location_to_j_id(item, location)
|
||||
services.ensure_visual_hashes(item)
|
||||
|
||||
update_fields = []
|
||||
if item.uploaded_by_id is None and task.user_id is not None:
|
||||
@@ -120,3 +144,7 @@ def run_download_task(task_id):
|
||||
speed=None,
|
||||
updated_at=timezone.now(),
|
||||
)
|
||||
finally:
|
||||
# Background threads hold their own DB connection; release it so
|
||||
# Gunicorn workers do not leak connections when threads finish.
|
||||
connection.close()
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
from django.core.management.base import BaseCommand
|
||||
from django.db.models import Q
|
||||
|
||||
from apps.library.models import MediaItem
|
||||
from apps.library.services import HASH_FIELDS, ensure_visual_hashes
|
||||
|
||||
|
||||
class Command(BaseCommand):
|
||||
help = "Compute perceptual hashes for library items missing them."
|
||||
|
||||
def add_arguments(self, parser):
|
||||
parser.add_argument(
|
||||
"--force",
|
||||
action="store_true",
|
||||
help="Recompute hashes even when they already exist.",
|
||||
)
|
||||
|
||||
def handle(self, *args, **options):
|
||||
queryset = MediaItem.objects.prefetch_related("locations").order_by("id")
|
||||
if not options["force"]:
|
||||
missing = Q()
|
||||
for field in HASH_FIELDS:
|
||||
missing |= Q(**{field: ""})
|
||||
queryset = queryset.filter(missing)
|
||||
|
||||
total = queryset.count()
|
||||
if total == 0:
|
||||
self.stdout.write(self.style.SUCCESS("All items already have hashes."))
|
||||
return
|
||||
|
||||
self.stdout.write(f"Computing hashes for {total} item(s) ...")
|
||||
done = 0
|
||||
for item in queryset.iterator(chunk_size=100):
|
||||
if options["force"]:
|
||||
item.ahash = item.dhash = item.phash = item.whash = ""
|
||||
ensure_visual_hashes(item)
|
||||
done += 1
|
||||
if done % 100 == 0:
|
||||
self.stdout.write(f"Processed {done} ...")
|
||||
|
||||
self.stdout.write(
|
||||
self.style.SUCCESS(f"Done. Processed {done} item(s).")
|
||||
)
|
||||
+33
@@ -0,0 +1,33 @@
|
||||
# Generated by Django 6.1.1 on 2026-09-17 17:34
|
||||
|
||||
from django.db import migrations, models
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
|
||||
dependencies = [
|
||||
('library', '0005_downloadtask'),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.AddField(
|
||||
model_name='mediaitem',
|
||||
name='ahash',
|
||||
field=models.CharField(blank=True, db_index=True, default='', max_length=32),
|
||||
),
|
||||
migrations.AddField(
|
||||
model_name='mediaitem',
|
||||
name='dhash',
|
||||
field=models.CharField(blank=True, db_index=True, default='', max_length=32),
|
||||
),
|
||||
migrations.AddField(
|
||||
model_name='mediaitem',
|
||||
name='phash',
|
||||
field=models.CharField(blank=True, db_index=True, default='', max_length=32),
|
||||
),
|
||||
migrations.AddField(
|
||||
model_name='mediaitem',
|
||||
name='whash',
|
||||
field=models.CharField(blank=True, db_index=True, default='', max_length=32),
|
||||
),
|
||||
]
|
||||
@@ -0,0 +1,18 @@
|
||||
# Generated by Django 6.1.1 on 2026-09-17 17:42
|
||||
|
||||
from django.db import migrations, models
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
|
||||
dependencies = [
|
||||
('library', '0006_mediaitem_ahash_mediaitem_dhash_mediaitem_phash_and_more'),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.AddField(
|
||||
model_name='tempupload',
|
||||
name='visual_matches',
|
||||
field=models.JSONField(blank=True, null=True),
|
||||
),
|
||||
]
|
||||
@@ -24,6 +24,11 @@ class MediaItem(models.Model):
|
||||
hidden_from_guests = models.BooleanField(default=False, db_index=True)
|
||||
e621_post_id = models.IntegerField(null=True, blank=True, db_index=True)
|
||||
e621_data = models.JSONField(null=True, blank=True)
|
||||
# Perceptual hashes (hex strings) used by the duplicates engine.
|
||||
ahash = models.CharField(max_length=32, blank=True, default="", db_index=True)
|
||||
dhash = models.CharField(max_length=32, blank=True, default="", db_index=True)
|
||||
phash = models.CharField(max_length=32, blank=True, default="", db_index=True)
|
||||
whash = models.CharField(max_length=32, blank=True, default="", db_index=True)
|
||||
created_at = models.DateTimeField(auto_now_add=True)
|
||||
updated_at = models.DateTimeField(auto_now=True)
|
||||
|
||||
@@ -100,6 +105,7 @@ class TempUpload(models.Model):
|
||||
custom_tags = models.JSONField(default=list, blank=True)
|
||||
custom_notes = models.TextField(blank=True, default="")
|
||||
iqdb_data = models.JSONField(null=True, blank=True)
|
||||
visual_matches = models.JSONField(null=True, blank=True)
|
||||
library_item = models.ForeignKey(
|
||||
MediaItem,
|
||||
null=True,
|
||||
|
||||
@@ -117,6 +117,7 @@ class TempUploadSerializer(serializers.ModelSerializer):
|
||||
"custom_tags",
|
||||
"custom_notes",
|
||||
"iqdb_data",
|
||||
"visual_matches",
|
||||
"library_j_id",
|
||||
"file_url",
|
||||
"preview_url",
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import mimetypes
|
||||
import os
|
||||
import re
|
||||
@@ -7,12 +8,19 @@ import shutil
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
import imagehash
|
||||
from django.conf import settings
|
||||
from django.http import FileResponse, Http404, HttpResponse
|
||||
from django.utils.text import get_valid_filename
|
||||
from PIL import Image
|
||||
|
||||
from .models import MediaItem, MediaLocation
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
HASH_FIELDS = ("ahash", "dhash", "phash", "whash")
|
||||
IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".gif", ".apng", ".webp"}
|
||||
|
||||
ALLOWED_EXTENSIONS = {
|
||||
".jpg",
|
||||
".jpeg",
|
||||
@@ -82,6 +90,41 @@ def rename_location_to_j_id(item, location):
|
||||
return location
|
||||
|
||||
|
||||
def compute_visual_hashes(path):
|
||||
"""Perceptual hashes for an image file (empty dict for other files)."""
|
||||
path = Path(path)
|
||||
if path.suffix.lower() not in IMAGE_EXTENSIONS:
|
||||
return {}
|
||||
try:
|
||||
with Image.open(path) as image:
|
||||
converted = image.convert("RGB")
|
||||
return {
|
||||
"ahash": str(imagehash.average_hash(converted, hash_size=8)),
|
||||
"dhash": str(imagehash.dhash(converted, hash_size=8)),
|
||||
"phash": str(imagehash.phash(converted, hash_size=8)),
|
||||
"whash": str(imagehash.whash(converted, hash_size=8)),
|
||||
}
|
||||
except Exception: # noqa: BLE001 - hashing must never break indexing
|
||||
logger.exception("Could not compute visual hashes for %s", path)
|
||||
return {}
|
||||
|
||||
|
||||
def ensure_visual_hashes(item):
|
||||
"""Fill in missing perceptual hashes for a media item."""
|
||||
if all(getattr(item, field) for field in HASH_FIELDS):
|
||||
return item
|
||||
location = item.locations.first()
|
||||
if location is None:
|
||||
return item
|
||||
hashes = compute_visual_hashes(location.path)
|
||||
if not hashes:
|
||||
return item
|
||||
for field, value in hashes.items():
|
||||
setattr(item, field, value)
|
||||
item.save(update_fields=[*hashes.keys(), "updated_at"])
|
||||
return item
|
||||
|
||||
|
||||
def parse_tags(raw):
|
||||
"""Normalize a comma-separated string or JSON list into a list of tags."""
|
||||
if raw is None:
|
||||
|
||||
@@ -0,0 +1,394 @@
|
||||
"""Library tooling: duplicates, deletion and storage overview."""
|
||||
|
||||
import logging
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
from django.conf import settings
|
||||
from django.core.cache import cache
|
||||
from django.db.models import Count, Q
|
||||
from rest_framework import status
|
||||
from rest_framework.permissions import IsAuthenticated
|
||||
from rest_framework.response import Response
|
||||
from rest_framework.views import APIView
|
||||
|
||||
from .models import MediaItem, MediaLocation
|
||||
from .permissions import CanUpload
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
HASH_FIELDS = ("ahash", "dhash", "phash", "whash")
|
||||
STORAGE_CACHE_KEY = "j621.library.storage"
|
||||
STORAGE_CACHE_TTL = 60
|
||||
GROUPS_PER_PAGE = 20
|
||||
|
||||
|
||||
def parse_threshold(value, default=0.8):
|
||||
try:
|
||||
parsed = float(value)
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
if parsed > 1:
|
||||
parsed = parsed / 100.0
|
||||
return min(max(parsed, 0.0), 1.0)
|
||||
|
||||
|
||||
def parse_algorithms(value):
|
||||
if not isinstance(value, list):
|
||||
return list(HASH_FIELDS)
|
||||
selected = [field for field in value if field in HASH_FIELDS]
|
||||
return selected or list(HASH_FIELDS)
|
||||
|
||||
|
||||
def hashes_similarity(first_hashes, second_hashes, algorithms, threshold):
|
||||
"""Best similarity between two hash mappings, or None below threshold."""
|
||||
best = None
|
||||
for field in algorithms:
|
||||
first = first_hashes.get(field) or ""
|
||||
second = second_hashes.get(field) or ""
|
||||
if not first or not second:
|
||||
continue
|
||||
try:
|
||||
distance = bin(int(first, 16) ^ int(second, 16)).count("1")
|
||||
except ValueError:
|
||||
continue
|
||||
value = 1.0 - distance / 64.0
|
||||
if best is None or value > best:
|
||||
best = value
|
||||
if best is None or best < threshold:
|
||||
return None
|
||||
return best
|
||||
|
||||
|
||||
def similarity_between(item_a, item_b, algorithms, threshold):
|
||||
"""Best similarity across the selected algorithms, or None below threshold."""
|
||||
return hashes_similarity(
|
||||
{field: getattr(item_a, field, "") for field in algorithms},
|
||||
{field: getattr(item_b, field, "") for field in algorithms},
|
||||
algorithms,
|
||||
threshold,
|
||||
)
|
||||
|
||||
|
||||
def hashed_items(algorithms):
|
||||
query = Q()
|
||||
for field in algorithms:
|
||||
query |= ~Q(**{field: ""})
|
||||
return list(MediaItem.objects.filter(query).prefetch_related("locations"))
|
||||
|
||||
|
||||
def display_rating(item):
|
||||
if item.rating:
|
||||
return item.rating
|
||||
data = item.e621_data or {}
|
||||
rating = data.get("rating") if isinstance(data, dict) else None
|
||||
return rating if rating in {"s", "q", "e"} else ""
|
||||
|
||||
|
||||
def item_brief(item):
|
||||
locations = list(item.locations.all())
|
||||
location = locations[0] if locations else None
|
||||
return {
|
||||
"j_id": f"J-{item.id}",
|
||||
"md5": item.md5,
|
||||
"filename": Path(location.rel_path).name if location else item.md5,
|
||||
"size": item.size,
|
||||
"rating": display_rating(item),
|
||||
"location_count": len(locations),
|
||||
"uploaded_by": item.uploaded_by.username if item.uploaded_by else None,
|
||||
"e621_post_id": item.e621_post_id,
|
||||
}
|
||||
|
||||
|
||||
def resolve_item(data):
|
||||
j_id = str(data.get("j_id") or "").strip()
|
||||
md5 = str(data.get("md5") or "").strip().lower()
|
||||
if j_id:
|
||||
numeric = j_id[2:] if j_id.upper().startswith("J-") else j_id
|
||||
if numeric.isdigit():
|
||||
item = MediaItem.objects.filter(pk=int(numeric)).first()
|
||||
if item is not None:
|
||||
return item
|
||||
if md5:
|
||||
return MediaItem.objects.filter(md5=md5).first()
|
||||
return None
|
||||
|
||||
|
||||
class ExactDuplicatesView(APIView):
|
||||
"""Items whose content exists at more than one path."""
|
||||
|
||||
permission_classes = [IsAuthenticated]
|
||||
|
||||
def get(self, request):
|
||||
items = (
|
||||
MediaItem.objects.annotate(location_count=Count("locations"))
|
||||
.filter(location_count__gt=1)
|
||||
.prefetch_related("locations")
|
||||
.order_by("-location_count", "id")
|
||||
)
|
||||
groups = []
|
||||
for item in items:
|
||||
brief = item_brief(item)
|
||||
brief["locations"] = [
|
||||
{"id": location.id, "rel_path": location.rel_path}
|
||||
for location in item.locations.all()
|
||||
]
|
||||
groups.append(brief)
|
||||
return Response({"count": len(groups), "groups": groups})
|
||||
|
||||
|
||||
class VisualMatchesView(APIView):
|
||||
"""Items visually similar to one library item."""
|
||||
|
||||
permission_classes = [IsAuthenticated]
|
||||
|
||||
def post(self, request):
|
||||
threshold = parse_threshold(request.data.get("threshold"))
|
||||
algorithms = parse_algorithms(request.data.get("algorithms"))
|
||||
target = resolve_item(request.data)
|
||||
if target is None:
|
||||
return Response(
|
||||
{"detail": "A j_id or md5 is required."},
|
||||
status=status.HTTP_400_BAD_REQUEST,
|
||||
)
|
||||
|
||||
matches = []
|
||||
for item in hashed_items(algorithms):
|
||||
if item.pk == target.pk:
|
||||
continue
|
||||
similarity = similarity_between(target, item, algorithms, threshold)
|
||||
if similarity is None:
|
||||
continue
|
||||
brief = item_brief(item)
|
||||
brief["similarity"] = round(similarity * 100, 1)
|
||||
matches.append(brief)
|
||||
matches.sort(key=lambda entry: entry["similarity"], reverse=True)
|
||||
|
||||
return Response(
|
||||
{
|
||||
"target": item_brief(target),
|
||||
"threshold": round(threshold * 100, 1),
|
||||
"algorithms": algorithms,
|
||||
"count": len(matches),
|
||||
"matches": matches[:200],
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
class VisualGroupsView(APIView):
|
||||
"""Groups of visually similar items (union-find over perceptual hashes).
|
||||
|
||||
Pairwise comparison is O(n^2) with fast bit operations, which is fine for
|
||||
a personal library. Revisit with a bucketed index if libraries grow huge.
|
||||
"""
|
||||
|
||||
permission_classes = [IsAuthenticated]
|
||||
|
||||
def post(self, request):
|
||||
threshold = parse_threshold(request.data.get("threshold"))
|
||||
algorithms = parse_algorithms(request.data.get("algorithms"))
|
||||
page = max(1, int(request.data.get("page") or 1))
|
||||
|
||||
items = hashed_items(algorithms)
|
||||
parent = list(range(len(items)))
|
||||
|
||||
def find(index):
|
||||
while parent[index] != index:
|
||||
parent[index] = parent[parent[index]]
|
||||
index = parent[index]
|
||||
return index
|
||||
|
||||
def union(first, second):
|
||||
root_a = find(first)
|
||||
root_b = find(second)
|
||||
if root_a != root_b:
|
||||
parent[root_b] = root_a
|
||||
|
||||
for first in range(len(items)):
|
||||
for second in range(first + 1, len(items)):
|
||||
if find(first) == find(second):
|
||||
continue
|
||||
if (
|
||||
similarity_between(
|
||||
items[first], items[second], algorithms, threshold
|
||||
)
|
||||
is not None
|
||||
):
|
||||
union(first, second)
|
||||
|
||||
grouped = {}
|
||||
for index, item in enumerate(items):
|
||||
grouped.setdefault(find(index), []).append(item)
|
||||
|
||||
groups = [members for members in grouped.values() if len(members) >= 2]
|
||||
groups.sort(key=len, reverse=True)
|
||||
|
||||
total = len(groups)
|
||||
start = (page - 1) * GROUPS_PER_PAGE
|
||||
page_groups = groups[start : start + GROUPS_PER_PAGE]
|
||||
|
||||
return Response(
|
||||
{
|
||||
"count": total,
|
||||
"page": page,
|
||||
"per_page": GROUPS_PER_PAGE,
|
||||
"has_next": start + GROUPS_PER_PAGE < total,
|
||||
"threshold": round(threshold * 100, 1),
|
||||
"algorithms": algorithms,
|
||||
"groups": [
|
||||
{"size": len(members), "members": [item_brief(item) for item in members]}
|
||||
for members in page_groups
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def can_delete(user, item):
|
||||
if user.is_superuser or user.role == user.ROLE_STAFF:
|
||||
return True
|
||||
return item.uploaded_by_id == user.id
|
||||
|
||||
|
||||
def remove_watched_file(path):
|
||||
"""Delete a file only when it lives inside the watched folder."""
|
||||
watched = Path(settings.WATCHED_FOLDER).resolve()
|
||||
try:
|
||||
resolved = Path(path).resolve()
|
||||
resolved.relative_to(watched)
|
||||
except (ValueError, OSError):
|
||||
return False
|
||||
if resolved.is_file():
|
||||
resolved.unlink()
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
class DeleteFilesView(APIView):
|
||||
"""Delete items (with every copy) or individual duplicate locations."""
|
||||
|
||||
permission_classes = [CanUpload]
|
||||
|
||||
def post(self, request):
|
||||
j_ids = request.data.get("j_ids") or []
|
||||
location_ids = request.data.get("location_ids") or []
|
||||
if not isinstance(j_ids, list) or not isinstance(location_ids, list):
|
||||
return Response(
|
||||
{"detail": "j_ids and location_ids must be lists."},
|
||||
status=status.HTTP_400_BAD_REQUEST,
|
||||
)
|
||||
|
||||
deleted = []
|
||||
errors = []
|
||||
|
||||
numeric_ids = []
|
||||
for value in j_ids:
|
||||
text = str(value).strip()
|
||||
numeric = text[2:] if text.upper().startswith("J-") else text
|
||||
if numeric.isdigit():
|
||||
numeric_ids.append(int(numeric))
|
||||
|
||||
for item in MediaItem.objects.filter(pk__in=numeric_ids).prefetch_related(
|
||||
"locations"
|
||||
):
|
||||
if not can_delete(request.user, item):
|
||||
errors.append({"j_id": f"J-{item.id}", "error": "permission denied"})
|
||||
continue
|
||||
for location in item.locations.all():
|
||||
remove_watched_file(location.path)
|
||||
deleted.append(f"J-{item.id}")
|
||||
item.delete()
|
||||
|
||||
cache.delete(STORAGE_CACHE_KEY)
|
||||
|
||||
for location in MediaLocation.objects.filter(
|
||||
id__in=location_ids
|
||||
).select_related("item"):
|
||||
if not can_delete(request.user, location.item):
|
||||
errors.append(
|
||||
{"location": location.id, "error": "permission denied"}
|
||||
)
|
||||
continue
|
||||
location_id = location.id
|
||||
item = location.item
|
||||
remove_watched_file(location.path)
|
||||
location.delete()
|
||||
if item.locations.exists():
|
||||
deleted.append(f"location {location_id}")
|
||||
else:
|
||||
deleted.append(f"J-{item.id}")
|
||||
item.delete()
|
||||
|
||||
return Response({"deleted": deleted, "errors": errors})
|
||||
|
||||
|
||||
class ClearTempView(APIView):
|
||||
"""Remove staged files from the temp upload folder."""
|
||||
|
||||
permission_classes = [CanUpload]
|
||||
|
||||
def post(self, request):
|
||||
temp_dir = Path(settings.MEDIA_ROOT) / "uploads" / "temp"
|
||||
removed = 0
|
||||
if temp_dir.exists():
|
||||
for entry in temp_dir.iterdir():
|
||||
if entry.is_file():
|
||||
entry.unlink()
|
||||
removed += 1
|
||||
cache.delete(STORAGE_CACHE_KEY)
|
||||
return Response({"removed": removed})
|
||||
|
||||
|
||||
class StorageView(APIView):
|
||||
"""Disk usage for the watched folder, media root and temp uploads."""
|
||||
|
||||
permission_classes = [IsAuthenticated]
|
||||
|
||||
def get(self, request):
|
||||
cached = cache.get(STORAGE_CACHE_KEY)
|
||||
if cached is not None:
|
||||
return Response(cached)
|
||||
|
||||
watched = Path(settings.WATCHED_FOLDER)
|
||||
media_root = Path(settings.MEDIA_ROOT)
|
||||
temp_dir = media_root / "uploads" / "temp"
|
||||
|
||||
usage = shutil.disk_usage(
|
||||
str(watched) if watched.exists() else str(Path(settings.BASE_DIR))
|
||||
)
|
||||
library_size = (
|
||||
sum(file.stat().st_size for file in watched.rglob("*") if file.is_file())
|
||||
if watched.exists()
|
||||
else 0
|
||||
)
|
||||
media_size = (
|
||||
sum(file.stat().st_size for file in media_root.rglob("*") if file.is_file())
|
||||
if media_root.exists()
|
||||
else 0
|
||||
)
|
||||
temp_files = (
|
||||
[file for file in temp_dir.rglob("*") if file.is_file()]
|
||||
if temp_dir.exists()
|
||||
else []
|
||||
)
|
||||
|
||||
payload = {
|
||||
"watched_folder": {
|
||||
"path": str(watched),
|
||||
"total": usage.total,
|
||||
"used": usage.used,
|
||||
"free": usage.free,
|
||||
"library_size": library_size,
|
||||
"percent_used": (
|
||||
round(usage.used / usage.total * 100, 1) if usage.total else 0
|
||||
),
|
||||
},
|
||||
"media": {"path": str(media_root), "size": media_size},
|
||||
"temp": {
|
||||
"path": str(temp_dir),
|
||||
"size": sum(file.stat().st_size for file in temp_files),
|
||||
"files": len(temp_files),
|
||||
},
|
||||
"library_items": MediaItem.objects.count(),
|
||||
}
|
||||
cache.set(STORAGE_CACHE_KEY, payload, STORAGE_CACHE_TTL)
|
||||
return Response(payload)
|
||||
@@ -28,10 +28,40 @@ from . import services
|
||||
from .models import MediaItem, TempUpload
|
||||
from .permissions import CanUpload
|
||||
from .serializers import TempUploadSerializer
|
||||
from .tools import HASH_FIELDS, hashes_similarity
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def find_library_matches(path, limit=10):
|
||||
"""Library items visually similar to a staged file."""
|
||||
hashes = services.compute_visual_hashes(path)
|
||||
if not hashes:
|
||||
return []
|
||||
algorithms = list(HASH_FIELDS)
|
||||
threshold = settings.VISUAL_MATCH_THRESHOLD
|
||||
matches = []
|
||||
for item in MediaItem.objects.prefetch_related("locations"):
|
||||
similarity = hashes_similarity(
|
||||
hashes,
|
||||
{field: getattr(item, field, "") for field in algorithms},
|
||||
algorithms,
|
||||
threshold,
|
||||
)
|
||||
if similarity is None:
|
||||
continue
|
||||
location = item.locations.first()
|
||||
matches.append(
|
||||
{
|
||||
"j_id": f"J-{item.id}",
|
||||
"filename": Path(location.rel_path).name if location else item.md5,
|
||||
"similarity": round(similarity * 100, 1),
|
||||
}
|
||||
)
|
||||
matches.sort(key=lambda entry: entry["similarity"], reverse=True)
|
||||
return matches[:limit]
|
||||
|
||||
|
||||
def complete_temp_upload(temp, download_url=None):
|
||||
"""Index the upload into the library.
|
||||
|
||||
@@ -57,6 +87,7 @@ def complete_temp_upload(temp, download_url=None):
|
||||
raise
|
||||
item, _, location, _ = services.index_file(destination, folder)
|
||||
services.rename_location_to_j_id(item, location)
|
||||
services.ensure_visual_hashes(item)
|
||||
if temp.file:
|
||||
temp.file.delete(save=False)
|
||||
else:
|
||||
@@ -69,6 +100,7 @@ def complete_temp_upload(temp, download_url=None):
|
||||
shutil.copyfileobj(source, target)
|
||||
item, _, location, _ = services.index_file(destination, folder)
|
||||
services.rename_location_to_j_id(item, location)
|
||||
services.ensure_visual_hashes(item)
|
||||
temp.file.delete(save=False)
|
||||
|
||||
temp.library_item = item
|
||||
@@ -143,6 +175,11 @@ class TempUploadViewSet(
|
||||
temp.resolution = TempUpload.RESOLUTION_DUPLICATE
|
||||
temp.library_item = existing
|
||||
temp.file.delete(save=False)
|
||||
else:
|
||||
matches = find_library_matches(temp.file.path)
|
||||
if matches:
|
||||
temp.visual_matches = matches
|
||||
temp.status = TempUpload.STATUS_VISUAL_MATCH
|
||||
temp.save()
|
||||
return Response(
|
||||
self.get_serializer(temp).data, status=status.HTTP_201_CREATED
|
||||
|
||||
@@ -1,6 +1,14 @@
|
||||
from django.urls import include, path
|
||||
from rest_framework.routers import DefaultRouter
|
||||
|
||||
from .tools import (
|
||||
ClearTempView,
|
||||
DeleteFilesView,
|
||||
ExactDuplicatesView,
|
||||
StorageView,
|
||||
VisualGroupsView,
|
||||
VisualMatchesView,
|
||||
)
|
||||
from .uploads import TempUploadViewSet
|
||||
from .views import ClientDownloadView, DownloadTaskViewSet, MediaItemViewSet
|
||||
|
||||
@@ -12,4 +20,22 @@ router.register("online/downloads", DownloadTaskViewSet, basename="download")
|
||||
urlpatterns = [
|
||||
path("", include(router.urls)),
|
||||
path("online/file/", ClientDownloadView.as_view(), name="client_download"),
|
||||
path(
|
||||
"duplicates/md5/",
|
||||
ExactDuplicatesView.as_view(),
|
||||
name="duplicates_md5",
|
||||
),
|
||||
path(
|
||||
"duplicates/visual/",
|
||||
VisualMatchesView.as_view(),
|
||||
name="duplicates_visual",
|
||||
),
|
||||
path(
|
||||
"duplicates/visual_groups/",
|
||||
VisualGroupsView.as_view(),
|
||||
name="duplicates_groups",
|
||||
),
|
||||
path("delete/", DeleteFilesView.as_view(), name="delete_files"),
|
||||
path("temp/clear/", ClearTempView.as_view(), name="clear_temp"),
|
||||
path("storage/", StorageView.as_view(), name="storage_info"),
|
||||
]
|
||||
|
||||
@@ -16,7 +16,7 @@ from rest_framework.response import Response
|
||||
from rest_framework.views import APIView
|
||||
|
||||
from . import services
|
||||
from .downloads import start_download_task
|
||||
from .downloads import reap_stale_downloads, start_download_task
|
||||
from .models import DownloadTask, MediaItem
|
||||
from .permissions import CanUpload, IsUploaderOrStaffOrReadOnly
|
||||
from .serializers import DownloadTaskSerializer, MediaItemSerializer
|
||||
@@ -41,6 +41,15 @@ class MediaItemViewSet(
|
||||
)
|
||||
if not self.request.user.is_authenticated:
|
||||
queryset = queryset.filter(hidden_from_guests=False)
|
||||
j_ids = self.request.query_params.get("j_ids", "").strip()
|
||||
if j_ids:
|
||||
numeric_ids = []
|
||||
for value in j_ids.split(","):
|
||||
text = value.strip()
|
||||
number = text[2:] if text.upper().startswith("J-") else text
|
||||
if number.isdigit():
|
||||
numeric_ids.append(int(number))
|
||||
queryset = queryset.filter(pk__in=numeric_ids)
|
||||
search = self.request.query_params.get("search", "").strip()
|
||||
if search:
|
||||
queryset = queryset.filter(locations__rel_path__icontains=search)
|
||||
@@ -203,6 +212,7 @@ class DownloadTaskViewSet(
|
||||
http_method_names = ["get", "post", "head", "options"]
|
||||
|
||||
def get_queryset(self):
|
||||
reap_stale_downloads()
|
||||
queryset = DownloadTask.objects.select_related("library_item")
|
||||
user = self.request.user
|
||||
if not (user.is_staff or user.is_superuser):
|
||||
|
||||
Reference in New Issue
Block a user