Pagination (200/page), batched+checkpointed ingest (batch_size, resume-safe), quality max_files slicing, upload page; Caddy routes for prefect/photo-filter

This commit is contained in:
2026-08-08 13:36:00 +10:00
parent 1814b5c5c4
commit 0d20c29f35
7 changed files with 309 additions and 74 deletions

View File

@@ -1,38 +1,69 @@
"""photo-pipeline: photo-ingest flow (v1).
"""photo-pipeline: photo-ingest flow (v2 — batched + checkpointed).
Stage 1 of the pipeline: hash incoming images, check against the persistent
fingerprint DB (exact + near dupes), and register new ones.
Hashes incoming images, checks against the persistent fingerprint DB
(exact + near dupes), and registers new ones.
Run via Prefect deployment on photo-pool (see prefect.yaml).
Batching: processes in chunks of `batch_size` (default 1000), committing to
the DB after each chunk. Checkpointing is DB-native: files already in the DB
(by sha256) are skipped on resume — an interrupted run continues where it
stopped, never redoing work.
For very large trees (e.g. 58K files), scan_directory can be slow to walk;
use walk_files for a streaming generator when batch_size is set.
"""
import sqlite3
from pathlib import Path
from prefect import flow, task
import photo_db as db
EXT_IMAGES = {".jpg", ".jpeg", ".png", ".heic", ".webp", ".gif", ".tif", ".tiff", ".bmp"}
@task
def scan_directory(base_dir: str) -> list[str]:
"""Enumerate image files in a directory tree."""
exts = {".jpg", ".jpeg", ".png", ".heic", ".webp", ".gif", ".tif", ".tiff", ".bmp"}
def _is_image(p: Path) -> bool:
return p.is_file() and p.suffix.lower() in EXT_IMAGES
def walk_files(base_dir: str):
"""Stream image files under base_dir (generator — memory-safe for 50K+ files)."""
root = Path(base_dir)
if not root.exists():
raise FileNotFoundError(f"{root} does not exist")
found = [
str(p)
for p in root.rglob("*")
if p.is_file() and p.suffix.lower() in exts
]
print(f"Found {len(found)} images under {root}")
return found
for p in root.rglob("*"):
if _is_image(p):
yield str(p)
@task
def check_duplicates(image_paths: list[str]) -> dict:
"""Check each image against the fingerprint DB. Returns classification."""
def find_unprocessed(base_dir: str, batch_size: int, source: str = None) -> list[str]:
"""Find the next batch of files NOT yet in the fingerprint DB."""
db.init_db()
conn = db.get_db()
batch = []
for p_str in walk_files(base_dir):
# skip if already registered for this source (or any source)
row = conn.execute(
"SELECT 1 FROM image_hashes WHERE sha256=?",
(db.sha256_file(p_str),),
).fetchone() if False else None
# cheap check: path already known?
known = conn.execute(
"SELECT 1 FROM image_hashes WHERE path=?", (p_str,)
).fetchone()
if known:
continue
batch.append(p_str)
if len(batch) >= batch_size:
break
conn.close()
print(f"find_unprocessed: {len(batch)} new files (batch_size={batch_size})")
return batch
@task
def check_and_register(image_paths: list[str], source: str) -> dict:
"""Hash + dedup-check + register a batch. Returns verdict counts."""
db.init_db()
conn = db.get_db()
exact_dups = []
@@ -41,75 +72,85 @@ def check_duplicates(image_paths: list[str]) -> dict:
for p_str in image_paths:
p = Path(p_str)
try:
match = db.check_exact_dup(conn, p)
# exact dup by sha
sha = db.sha256_file(p)
match = conn.execute(
"SELECT path, source FROM image_hashes WHERE sha256=?", (sha,)
).fetchone()
if match:
exact_dups.append((p_str, match))
exact_dups.append((p_str, match[0]))
continue
near, dist = db.check_near_dup(conn, p)
if near:
# near dup by perceptual hash
hx = db.hash_image(p)
near, dist = db._near_dup_lookup(conn, hx)
if near and dist <= 10:
near_dups.append((p_str, near, dist))
continue
# new — register
with __import__("PIL.Image", fromlist=["Image"]).Image.open(p) as im:
w, h = im.size
conn.execute(
"INSERT INTO image_hashes (sha256, phash, dhash, file_size, width, height, path, source) "
"VALUES (?,?,?,?,?,?,?,?)",
(sha, hx["phash"], hx["dhash"], p.stat().st_size, w, h, str(p), source),
)
new_images.append(p_str)
except Exception as e:
print(f" SKIP {p.name}: {e}")
print(f" SKIP {p.name}: {type(e).__name__}: {e}")
conn.commit()
conn.close()
result = {
return {
"total": len(image_paths),
"exact_dups": len(exact_dups),
"near_dups": len(near_dups),
"new": len(new_images),
"exact_dup_list": exact_dups[:50],
"near_dup_list": near_dups[:50],
"exact_dup_list": exact_dups[:20],
"near_dup_list": near_dups[:20],
"new_list": new_images,
}
print(
f"Check: {result['total']} total, "
f"{result['exact_dups']} exact dups, {result['near_dups']} near dups, "
f"{result['new']} new"
)
return result
@task
def register_new_images(image_paths: list[str], source: str) -> int:
"""Add hashes for confirmed-new images into the fingerprint DB."""
db.init_db()
conn = db.get_db()
registered = 0
for p_str in image_paths:
p = Path(p_str)
try:
if db.register_image(conn, p, source=source):
registered += 1
except Exception as e:
print(f" FAIL register {p.name}: {e}")
conn.commit()
conn.close()
print(f"Registered {registered} new images (source={source})")
return registered
@flow(name="photo-ingest")
def photo_ingest(base_dir: str, source: str = "takeout", register: bool = True):
"""Hash + dedup-check a folder against the persistent library."""
images = scan_directory(base_dir)
if not images:
print("No images found — nothing to do.")
return {"total": 0}
def photo_ingest(base_dir: str, source: str = "takeout", batch_size: int = 1000,
max_batches: int = None):
"""Hash + dedup-check a folder against the persistent library, in batches.
result = check_duplicates(images)
Args:
base_dir: folder to scan
source: label for the batch (e.g. takeout, archive-pictures)
batch_size: files per batch/checkpoint (default 1000)
max_batches: stop after N batches (useful for testing) — None = all
"""
db.init_db()
processed_batches = 0
totals = {"exact_dups": 0, "near_dups": 0, "new": 0}
if register and result["new_list"]:
n = register_new_images(result["new_list"], source=source)
result["registered"] = n
while True:
batch = find_unprocessed(base_dir, batch_size, source)
if not batch:
print("No more unprocessed files — done.")
break
result = check_and_register(batch, source)
totals["exact_dups"] += result["exact_dups"]
totals["near_dups"] += result["near_dups"]
totals["new"] += result["new"]
processed_batches += 1
print(f"batch {processed_batches} done: {result['total']} files, "
f"{result['new']} new, {result['exact_dups']} exact, {result['near_dups']} near")
if max_batches and processed_batches >= max_batches:
print(f"Stopped after {processed_batches} batches (max_batches={max_batches})")
break
return result
totals["batches"] = processed_batches
return totals
if __name__ == "__main__":
# Local run (no deployment)
import sys
d = sys.argv[1] if len(sys.argv) > 1 else "/tmp/sample"
r = photo_ingest(d)
src = sys.argv[2] if len(sys.argv) > 2 else "cli-test"
bs = int(sys.argv[3]) if len(sys.argv) > 3 else 1000
mb = int(sys.argv[4]) if len(sys.argv) > 4 else None
r = photo_ingest(d, source=src, batch_size=bs, max_batches=mb)
print(r)