Library search upgrades: tag search, tag cloud, status filter

Backend:
- MediaItem gains search_tags (custom + e621 tags, lowercase) and
  has_custom_data, maintained on save with a data migration backfill
- File list search accepts search_type=filename|tags|both (tag search is
  word-AND across the flattened tag text) and status=matched|custom|
  unknown filters
- New /api/tags/cloud/ endpoint (cached 2 min per guest/auth, invalidated
  on item changes and deletions) returning the most-used tags, honouring
  guest visibility

Frontend:
- Library sidebar: Filename/Tags/Both selector, status pill toggles
  (persisted), and a clickable tag cloud that runs a tag search
- Roadmap updated
This commit is contained in:
2026-09-17 13:19:17 -05:00
parent 7ec8ee974e
commit 4df573da43
7 changed files with 286 additions and 8 deletions
@@ -0,0 +1,43 @@
# Generated by Django 6.1.1 on 2026-09-17 18:04
from django.db import migrations, models
def backfill_derived_fields(apps, schema_editor):
MediaItem = apps.get_model("library", "MediaItem")
for item in MediaItem.objects.all().iterator(chunk_size=200):
names = [str(tag) for tag in (item.tags or [])]
data = item.e621_data if isinstance(item.e621_data, dict) else {}
categories = data.get("tags")
if isinstance(categories, dict):
for values in categories.values():
if isinstance(values, list):
names.extend(str(tag) for tag in values)
search_tags = " ".join(
sorted({str(name).strip().lower() for name in names if str(name).strip()})
)
MediaItem.objects.filter(pk=item.pk).update(
search_tags=search_tags,
has_custom_data=bool(item.tags or item.notes),
)
class Migration(migrations.Migration):
dependencies = [
('library', '0007_tempupload_visual_matches'),
]
operations = [
migrations.AddField(
model_name='mediaitem',
name='has_custom_data',
field=models.BooleanField(db_index=True, default=False),
),
migrations.AddField(
model_name='mediaitem',
name='search_tags',
field=models.TextField(blank=True, default=''),
),
migrations.RunPython(backfill_derived_fields, migrations.RunPython.noop),
]
+33 -1
View File
@@ -4,6 +4,26 @@ from django.db import models
import uuid
TAG_CLOUD_CACHE_KEYS = [
"j621.library.tag_cloud.auth",
"j621.library.tag_cloud.guest",
]
def build_search_tags(tags, e621_data):
"""Lowercase tag text (custom + e621) used for library tag search."""
names = [str(tag) for tag in (tags or [])]
data = e621_data if isinstance(e621_data, dict) else {}
categories = data.get("tags")
if isinstance(categories, dict):
for values in categories.values():
if isinstance(values, list):
names.extend(str(tag) for tag in values)
return " ".join(
sorted({str(name).strip().lower() for name in names if str(name).strip()})
)
class MediaItem(models.Model):
"""A logical media file, identified by its MD5 fingerprint."""
@@ -29,6 +49,9 @@ class MediaItem(models.Model):
dhash = models.CharField(max_length=32, blank=True, default="", db_index=True)
phash = models.CharField(max_length=32, blank=True, default="", db_index=True)
whash = models.CharField(max_length=32, blank=True, default="", db_index=True)
# Derived search/filter helpers: flattened tag text and custom-data flag.
search_tags = models.TextField(blank=True, default="")
has_custom_data = models.BooleanField(default=False, db_index=True)
created_at = models.DateTimeField(auto_now_add=True)
updated_at = models.DateTimeField(auto_now=True)
@@ -39,13 +62,22 @@ class MediaItem(models.Model):
return f"J-{self.pk} ({self.md5})"
def save(self, *args, **kwargs):
from django.core.cache import cache
from .guest_filter import item_is_hidden_for_guests
self.hidden_from_guests = item_is_hidden_for_guests(self)
self.has_custom_data = bool(self.tags or self.notes)
self.search_tags = build_search_tags(self.tags, self.e621_data)
update_fields = kwargs.get("update_fields")
if update_fields is not None:
kwargs["update_fields"] = set(update_fields) | {"hidden_from_guests"}
kwargs["update_fields"] = set(update_fields) | {
"hidden_from_guests",
"has_custom_data",
"search_tags",
}
super().save(*args, **kwargs)
cache.delete_many(TAG_CLOUD_CACHE_KEYS)
class MediaLocation(models.Model):
+53 -2
View File
@@ -8,11 +8,11 @@ from django.conf import settings
from django.core.cache import cache
from django.db.models import Count, Q
from rest_framework import status
from rest_framework.permissions import IsAuthenticated
from rest_framework.permissions import AllowAny, IsAuthenticated
from rest_framework.response import Response
from rest_framework.views import APIView
from .models import MediaItem, MediaLocation
from .models import TAG_CLOUD_CACHE_KEYS, MediaItem, MediaLocation
from .permissions import CanUpload
logger = logging.getLogger(__name__)
@@ -299,6 +299,7 @@ class DeleteFilesView(APIView):
item.delete()
cache.delete(STORAGE_CACHE_KEY)
cache.delete_many(TAG_CLOUD_CACHE_KEYS)
for location in MediaLocation.objects.filter(
id__in=location_ids
@@ -338,6 +339,56 @@ class ClearTempView(APIView):
return Response({"removed": removed})
class TagCloudView(APIView):
"""Most-used tags across the library (custom + e621 tags)."""
permission_classes = [AllowAny]
def get(self, request):
cache_key = (
"j621.library.tag_cloud.auth"
if request.user.is_authenticated
else "j621.library.tag_cloud.guest"
)
cached = cache.get(cache_key)
if cached is not None:
return Response(cached)
queryset = MediaItem.objects.all()
if not request.user.is_authenticated:
queryset = queryset.filter(hidden_from_guests=False)
counts = {}
for item in queryset.only("tags", "e621_data").iterator(chunk_size=500):
names = set()
for tag in item.tags or []:
name = str(tag).strip().lower()
if name:
names.add(name)
data = item.e621_data or {}
categories = data.get("tags") if isinstance(data, dict) else None
if isinstance(categories, dict):
for values in categories.values():
if isinstance(values, list):
names.update(
str(tag).strip().lower()
for tag in values
if str(tag).strip()
)
for name in names:
counts[name] = counts.get(name, 0) + 1
ranked = sorted(counts.items(), key=lambda entry: (-entry[1], entry[0]))
payload = {
"count": len(counts),
"tags": [
{"tag": tag, "count": count} for tag, count in ranked[:120]
],
}
cache.set(cache_key, payload, 120)
return Response(payload)
class StorageView(APIView):
"""Disk usage for the watched folder, media root and temp uploads."""
+2
View File
@@ -6,6 +6,7 @@ from .tools import (
DeleteFilesView,
ExactDuplicatesView,
StorageView,
TagCloudView,
VisualGroupsView,
VisualMatchesView,
)
@@ -37,5 +38,6 @@ urlpatterns = [
),
path("delete/", DeleteFilesView.as_view(), name="delete_files"),
path("temp/clear/", ClearTempView.as_view(), name="clear_temp"),
path("tags/cloud/", TagCloudView.as_view(), name="tag_cloud"),
path("storage/", StorageView.as_view(), name="storage_info"),
]
+33 -2
View File
@@ -4,7 +4,7 @@ from urllib.parse import urlparse
from django.conf import settings
from django.core import signing
from django.db.models import Min
from django.db.models import Min, Q
from django.http import Http404, StreamingHttpResponse
from django.shortcuts import get_object_or_404
from django.utils import timezone
@@ -51,8 +51,20 @@ class MediaItemViewSet(
numeric_ids.append(int(number))
queryset = queryset.filter(pk__in=numeric_ids)
search = self.request.query_params.get("search", "").strip()
search_type = self.request.query_params.get("search_type", "filename").strip()
if search:
queryset = queryset.filter(locations__rel_path__icontains=search)
words = [word for word in search.lower().split() if word]
if search_type == "tags":
for word in words:
queryset = queryset.filter(search_tags__icontains=word)
elif search_type == "both":
query = Q(locations__rel_path__icontains=search)
tag_query = Q()
for word in words:
tag_query &= Q(search_tags__icontains=word)
queryset = queryset.filter(query | tag_query)
else:
queryset = queryset.filter(locations__rel_path__icontains=search)
ratings = [
value
for value in self.request.query_params.get("rating", "").split(",")
@@ -60,6 +72,25 @@ class MediaItemViewSet(
]
if ratings:
queryset = queryset.filter(rating__in=ratings)
statuses = []
for value in self.request.query_params.getlist("status"):
statuses.extend(
part.strip() for part in value.split(",") if part.strip()
)
if statuses:
status_query = Q()
for status_value in set(statuses):
if status_value == "custom":
status_query |= Q(has_custom_data=True)
elif status_value == "matched":
status_query |= Q(
has_custom_data=False, e621_post_id__isnull=False
)
elif status_value == "unknown":
status_query |= Q(
has_custom_data=False, e621_post_id__isnull=True
)
queryset = queryset.filter(status_query)
ordering = self.request.query_params.get("ordering", "").strip()
queryset = queryset.order_by(
ordering if ordering in LIST_ORDERINGS else "-created_at"