Wait for all uploads, then batch MD5 -> visual -> IQDB with bulk links

Batching must not start while files are still being uploaded, and the MD5
phase must move a whole chunk at once instead of one resolve per file:

- the upload queue drains completely first (failed uploads included) before
  any matching starts;
- phase 1 asks e621 for every md5 (75 per posts.json request), builds the
  md5 -> post map from the response, and sends the matches to the new
  POST /api/uploads/link-bulk/ action, so a whole 75-file chunk moves into
  Indexed in a single board update;
- link-bulk indexes the staged file directly when the post's MD5 matches
  (identical bytes), so there is no per-file download round trip;
- phase 2 runs local visual similarity for whatever stayed pending, phase 3
  the IQDB queue.

Verified end to end with real e621 files: one md5 query for the batch, one
link-bulk call, both matching files flipped to Indexed together, then the
visual and IQDB phases. 23 library tests green (link-bulk, visual phase,
deferred visual matching).
This commit is contained in:
2026-09-19 11:10:24 -05:00
parent e2bf1c457f
commit 3a07481dfc
3 changed files with 211 additions and 49 deletions
+71 -46
View File
@@ -28,6 +28,7 @@ import {
fetchPostsByMd5,
iqdbSearch,
RATING_LABELS,
type E621Post,
} from "@/lib/e621";
import { formatBytes } from "@/lib/format";
import type { E621IqdbCandidate, Rating, TempUpload } from "@/lib/types";
@@ -857,12 +858,6 @@ export default function UploadPage() {
});
}
async function refreshUploads(): Promise<TempUpload[]> {
const fresh = await api<TempUpload[]>("/api/uploads/");
queryClient.setQueryData(["uploads"], fresh);
return fresh;
}
// One serial drain for all IQDB checks: clicking "Check similarity" many
// times used to start overlapping runs that re-downloaded the same staging
// blobs and piled requests onto the e621 queue until everything stalled.
@@ -962,61 +957,88 @@ export default function UploadPage() {
}
}
/** One link-bulk request for the batch, with a smaller-chunk fallback. */
async function bulkLink(
links: { temp_id: string; post: E621Post }[],
): Promise<TempUpload[]> {
try {
const result = await api<{ updated: TempUpload[] }>(
"/api/uploads/link-bulk/",
{ method: "POST", json: { links } },
);
return result.updated;
} catch (error) {
if (links.length <= 15) throw error;
// Very large files can make one batch request slow; retry in pieces.
const updated: TempUpload[] = [];
for (let index = 0; index < links.length; index += 15) {
const result = await api<{ updated: TempUpload[] }>(
"/api/uploads/link-bulk/",
{ method: "POST", json: { links: links.slice(index, index + 15) } },
);
updated.push(...result.updated);
}
return updated;
}
}
/**
* Phased processing for one finished upload batch: every file goes through
* the e621 MD5 lookup (batched), then local visual similarity, then IQDB.
* Phases never interleave per file, and each step lands on the board as it
* finishes.
* Phases for one finished upload batch, strictly in order: every file's
* e621 MD5 (one query + one bulk link call per 75 files, so a chunk moves
* into the library in a single board update), then every file's local
* visual similarity, then IQDB.
*/
async function runUploadPipeline(created: TempUpload[]) {
async function processUploads(created: TempUpload[]) {
const creds = effectiveCredentials(credentials);
const targets = created.filter((temp) => temp.status === "pending");
if (targets.length === 0) {
const pending = created.filter((temp) => temp.status === "pending");
if (pending.length === 0) {
invalidateUploads();
return;
}
const targetIds = new Set(targets.map((temp) => temp.temp_id));
const ids = new Set(pending.map((temp) => temp.temp_id));
// Phase 1: e621 MD5 lookup, 75 md5: metatags per query.
setPipeline({ phase: "md5", done: 0, total: targets.length });
// Phase 1: ask e621 for every md5, 75 per request; each chunk is then
// linked with a single bulk call so it flips to Indexed together.
setPipeline({ phase: "md5", done: 0, total: pending.length });
try {
for (let index = 0; index < targets.length; index += MD5_BATCH_SIZE) {
const md5Batch = targets.slice(index, index + MD5_BATCH_SIZE);
for (let index = 0; index < pending.length; index += MD5_BATCH_SIZE) {
const chunk = pending.slice(index, index + MD5_BATCH_SIZE);
const posts = await fetchPostsByMd5(
creds,
md5Batch.map((temp) => temp.md5),
chunk.map((temp) => temp.md5),
);
for (const post of posts) {
const match = md5Batch.find((temp) => temp.md5 === post.file.md5);
if (!match) continue;
try {
const updated = await api<TempUpload>(
`/api/uploads/${match.temp_id}/resolve/`,
{ method: "POST", json: { mode: "link", post, auto: true } },
const byMd5 = new Map(posts.map((post) => [post.file.md5, post]));
const links = chunk.flatMap((temp) => {
const post = temp.md5 ? byMd5.get(temp.md5) : undefined;
return post ? [{ temp_id: temp.temp_id, post }] : [];
});
if (links.length > 0) {
const updated = await bulkLink(links);
if (updated.length > 0) {
// One request, one board update: the chunk moves together.
const byId = new Map(updated.map((temp) => [temp.temp_id, temp]));
queryClient.setQueryData<TempUpload[]>(["uploads"], (current) =>
current
? current.map((item) => byId.get(item.temp_id) ?? item)
: current,
);
upsertUpload(updated);
} catch {
// Leave it pending.
}
}
setPipeline((current) =>
current
? {
...current,
done: Math.min(index + MD5_BATCH_SIZE, targets.length),
}
? { ...current, done: Math.min(index + MD5_BATCH_SIZE, pending.length) }
: current,
);
}
} catch {
// e621 unavailable; those files stay pending.
// e621 unavailable or a chunk failed; those files stay pending.
}
setPipeline(null);
// Phase 2: local visual similarity, one file at a time.
// Phase 2: local visual similarity for whatever is still pending.
const cached = queryClient.getQueryData<TempUpload[]>(["uploads"]) ?? [];
const visualTargets = cached.filter(
(temp) => targetIds.has(temp.temp_id) && temp.status === "pending",
(temp) => ids.has(temp.temp_id) && temp.status === "pending",
);
setPipeline({ phase: "visual", done: 0, total: visualTargets.length });
for (const temp of visualTargets) {
@@ -1035,10 +1057,12 @@ export default function UploadPage() {
}
setPipeline(null);
// Phase 3: IQDB for everything still unresolved (serial queue with its
// Phase 3: IQDB for the batch's unresolved files (serial queue with its
// own progress, cooldown and fatal-error stop).
const fresh = await refreshUploads();
checkSimilarity(unresolvedIqdbIds(fresh));
const remaining = (
queryClient.getQueryData<TempUpload[]>(["uploads"]) ?? []
).filter((temp) => ids.has(temp.temp_id));
checkSimilarity(unresolvedIqdbIds(remaining));
}
/** Mark an entry done, fade it out and drop it once it has been seen. */
@@ -1096,9 +1120,10 @@ export default function UploadPage() {
runningRef.current = false;
setBusy(false);
invalidateUploads();
// Batching only starts once every upload settled (success or failure).
if (created.length > 0) {
setProcessing(true);
void runUploadPipeline(created).finally(() => setProcessing(false));
void processUploads(created).finally(() => setProcessing(false));
}
}
@@ -1199,12 +1224,6 @@ export default function UploadPage() {
retry checks
</button>
</div>
) : busy ? (
<PhaseBar
label={`Uploading ${Math.min(settledCount + 1, batch.total)}/${batch.total}…`}
percent={batchPercent}
tone="bg-ctp-mauve"
/>
) : pipeline ? (
<PhaseBar
label={`${
@@ -1217,6 +1236,12 @@ export default function UploadPage() {
}
tone={pipeline.phase === "md5" ? "bg-ctp-peach" : "bg-ctp-lavender"}
/>
) : busy ? (
<PhaseBar
label={`Uploading ${Math.min(settledCount + 1, batch.total)}/${batch.total}…`}
percent={batchPercent}
tone="bg-ctp-mauve"
/>
) : processing && iqdbProgress ? (
<PhaseBar
label={`Checking IQDB — ${iqdbProgress.done}/${iqdbProgress.total}${