Pagination (200/page), batched+checkpointed ingest (batch_size, resume-safe), quality max_files slicing, upload page; Caddy routes for prefect/photo-filter

This commit is contained in:
2026-08-08 13:36:00 +10:00
parent 1814b5c5c4
commit 0d20c29f35
7 changed files with 309 additions and 74 deletions

View File

@@ -96,25 +96,34 @@ def index(request: Request):
@app.get("/review", response_class=HTMLResponse) @app.get("/review", response_class=HTMLResponse)
def review(request: Request, source: str = None, status: str = None): def review(request: Request, source: str = None, status: str = None, page: int = 1):
conn = _conn() conn = _conn()
per_page = 200
q = "SELECT sha256, path, status, source FROM image_hashes WHERE 1=1" q = "SELECT sha256, path, status, source FROM image_hashes WHERE 1=1"
count_q = "SELECT COUNT(*) FROM image_hashes WHERE 1=1"
params = [] params = []
if source: if source:
q += " AND source=?" q += " AND source=?"
count_q += " AND source=?"
params.append(source) params.append(source)
if status: if status:
q += " AND status=?" q += " AND status=?"
count_q += " AND status=?"
params.append(status) params.append(status)
else: else:
q += " AND status IN ('scanned','review')" q += " AND status IN ('scanned','review')"
q += " ORDER BY added_at DESC LIMIT 300" count_q += " AND status IN ('scanned','review')"
rows = conn.execute(q, params).fetchall() total = conn.execute(count_q, params).fetchone()[0]
pages = max(1, (total + per_page - 1) // per_page)
page = max(1, min(page, pages))
q += " ORDER BY added_at DESC LIMIT ? OFFSET ?"
rows = conn.execute(q, params + [per_page, (page - 1) * per_page]).fetchall()
conn.close() conn.close()
items = [_load_item(r) for r in rows] items = [_load_item(r) for r in rows]
return templates.TemplateResponse( return templates.TemplateResponse(
request, "review.html", request, "review.html",
{"items": items, "source": source, "status": status}, {"items": items, "source": source, "status": status,
"page": page, "pages": pages, "total": total},
) )
@@ -231,6 +240,47 @@ def thumb_file(name: str):
return FileResponse(f) return FileResponse(f)
@app.get("/upload", response_class=HTMLResponse)
def upload_page(request: Request):
"""Upload page — drop files/archives into the incoming folder."""
return templates.TemplateResponse(request, "upload.html", {})
@app.post("/upload")
async def upload(request: Request):
"""Receive uploaded files → save to /mnt/data/takeout/incoming/."""
import uuid
from starlette.datastructures import UploadFile
form = await request.form()
incoming = STAGING / "takeout" / "incoming"
incoming.mkdir(parents=True, exist_ok=True)
saved = []
for field in form.values():
if isinstance(field, UploadFile) and field.filename:
# sanitize: keep name but avoid path traversal
name = Path(field.filename).name
dest = incoming / f"{uuid.uuid4().hex[:8]}_{name}"
with open(dest, "wb") as f:
while chunk := await field.read(1024 * 1024):
f.write(chunk)
saved.append(dest.name)
# notify
try:
import sys
if str(BASE.parent) not in sys.path:
sys.path.insert(0, str(BASE.parent))
import apprise_helper
apprise_helper.notify(
"📥 photo-pipeline: upload received",
f"{len(saved)} file(s) saved to incoming. Watch flow will process them.",
)
except Exception:
pass
return {"saved": len(saved), "files": saved}
if __name__ == "__main__": if __name__ == "__main__":
import uvicorn import uvicorn

View File

@@ -41,6 +41,9 @@
.none { color: #666; font-style: italic; padding: 2rem; text-align: center; } .none { color: #666; font-style: italic; padding: 2rem; text-align: center; }
.selall { display: inline-flex; align-items: center; gap: .35rem; background: #222; border: 1px solid #444; border-radius: 6px; padding: .4rem .8rem; cursor: pointer; } .selall { display: inline-flex; align-items: center; gap: .35rem; background: #222; border: 1px solid #444; border-radius: 6px; padding: .4rem .8rem; cursor: pointer; }
.toast { position: fixed; bottom: 1.2rem; right: 1.2rem; background: #1d4; color: #031; padding: .7rem 1rem; border-radius: 8px; display: none; z-index: 50; font-size: .9rem; box-shadow: 0 2px 12px rgba(0,0,0,.5); } .toast { position: fixed; bottom: 1.2rem; right: 1.2rem; background: #1d4; color: #031; padding: .7rem 1rem; border-radius: 8px; display: none; z-index: 50; font-size: .9rem; box-shadow: 0 2px 12px rgba(0,0,0,.5); }
.pager { display: flex; gap: 1rem; align-items: center; justify-content: center; padding: 1.5rem; }
.pager a { color: #6cf; text-decoration: none; padding: .4rem .8rem; background: #222; border: 1px solid #444; border-radius: 6px; }
.pager span { color: #999; }
</style> </style>
</head> </head>
<body> <body>
@@ -72,7 +75,7 @@
<button class="btn-reset" id="bulk-reset">↺ Reset</button> <button class="btn-reset" id="bulk-reset">↺ Reset</button>
</div> </div>
<div class="toast" id="toast"></div> <div class="toast" id="toast"></div>
<p class="count">{{ items|length }} images</p> <p class="count">{{ total }} images · page {{ page }}/{{ pages }}</p>
<main> <main>
<div class="grid" id="review-grid"> <div class="grid" id="review-grid">
{% for item in items %} {% for item in items %}
@@ -81,6 +84,17 @@
<p class="none">No images match the filter.</p> <p class="none">No images match the filter.</p>
{% endfor %} {% endfor %}
</div> </div>
{% if pages > 1 %}
<div class="pager">
{% if page > 1 %}
<a href="/review?page={{ page-1 }}{% if source %}&source={{ source }}{% endif %}{% if status %}&status={{ status }}{% endif %}">← Prev</a>
{% endif %}
<span>page {{ page }} / {{ pages }}</span>
{% if page < pages %}
<a href="/review?page={{ page+1 }}{% if source %}&source={{ source }}{% endif %}{% if status %}&status={{ status }}{% endif %}">Next →</a>
{% endif %}
</div>
{% endif %}
</main> </main>
<script> <script>
// Selection is DERIVED from DOM checkboxes — single source of truth. // Selection is DERIVED from DOM checkboxes — single source of truth.
@@ -158,7 +172,11 @@
function updateCount() { function updateCount() {
const el = document.querySelector('.count'); const el = document.querySelector('.count');
const n = document.querySelectorAll('.card').length; const n = document.querySelectorAll('.card').length;
if (el) el.textContent = n + ' images'; if (el && el.textContent.includes('images')) {
// preserve the "· page X/Y" part, update the leading card count
const m = el.textContent.match(/page (\d+)\/(\d+)/);
el.textContent = n + ' images' + (m ? ' · page ' + m[1] + '/' + m[2] : '');
}
} }
function showToast(msg, isErr) { function showToast(msg, isErr) {

View File

@@ -0,0 +1,80 @@
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Upload — photo-pipeline</title>
<style>
:root { color-scheme: dark; }
body { font-family: system-ui, sans-serif; margin: 0; background: #111; color: #eee; }
header { padding: 1rem 1.5rem; border-bottom: 1px solid #333; display: flex; gap: 1.5rem; align-items: baseline; }
header h1 { font-size: 1.2rem; margin: 0; }
header a { color: #6cf; text-decoration: none; }
main { padding: 1.5rem; max-width: 700px; margin: 0 auto; }
.drop { border: 2px dashed #444; border-radius: 12px; padding: 3rem 2rem; text-align: center; color: #888; }
.drop.dragover { border-color: #6cf; background: #16222a; }
input[type=file] { margin: 1rem 0; }
button { background: #1d4; color: #031; border: 0; border-radius: 8px; padding: .6rem 1.2rem; font-size: 1rem; cursor: pointer; font-weight: 600; }
.hint { color: #666; font-size: .85rem; margin-top: 1.5rem; }
.hint code { background: #222; padding: .1rem .4rem; border-radius: 4px; }
</style>
</head>
<body>
<header>
<h1>📸 photo-pipeline</h1>
<a href="/">Overview</a>
<a href="/review">Review queue</a>
<a href="/pipeline">Pipeline</a>
<a href="/upload">Upload</a>
</header>
<main>
<h2>Upload to pipeline</h2>
<p>Drop Takeout archives (<code>.zip</code>/<code>.tgz</code>) or manifest files here.
They land in <code>/mnt/data/takeout/incoming/</code> — the watch flow picks them up
within 15 minutes and runs the pipeline automatically.</p>
<form id="upform" method="post" action="/upload" enctype="multipart/form-data">
<div class="drop" id="drop">
<p>Drag &amp; drop files here, or click to browse</p>
<input type="file" id="fileinput" name="files" multiple>
</div>
<p><button type="submit">Upload</button></p>
<p id="status" style="color:#6cf;"></p>
</form>
<div class="hint">
<p>What can go here:</p>
<ul>
<li><code>*.txt</code> — a manifest of Takeout download URLs (one per line). The pipeline downloads them itself.</li>
<li><code>*.zip</code> / <code>*.tgz</code> — an already-downloaded Takeout archive.</li>
</ul>
</div>
</main>
<script>
const drop = document.getElementById('drop');
const input = document.getElementById('fileinput');
const status = document.getElementById('status');
drop.addEventListener('dragover', e => { e.preventDefault(); drop.classList.add('dragover'); });
drop.addEventListener('dragleave', () => drop.classList.remove('dragover'));
drop.addEventListener('drop', e => {
e.preventDefault();
drop.classList.remove('dragover');
input.files = e.dataTransfer.files;
status.textContent = input.files.length + ' file(s) selected';
});
input.addEventListener('change', () => {
status.textContent = input.files.length + ' file(s) selected';
});
document.getElementById('upform').addEventListener('submit', async e => {
e.preventDefault();
status.textContent = 'Uploading…';
const fd = new FormData(document.getElementById('upform'));
const resp = await fetch('/upload', { method: 'POST', body: fd });
const data = await resp.json();
status.textContent = `Done: ${data.saved} file(s) saved to incoming. Watch flow will process them.`;
});
</script>
</body>
</html>

View File

@@ -161,3 +161,20 @@ def register_takeout_archive(conn, archive_name: str, path: Path) -> int:
if __name__ == "__main__": if __name__ == "__main__":
init_db() init_db()
print(f"DB ready at {DB_PATH}") print(f"DB ready at {DB_PATH}")
def _near_dup_lookup(conn, hx: dict, hamming_threshold: int = 10):
"""Find nearest perceptual-hash match, given precomputed hashes (avoids re-hash)."""
rows = conn.execute("SELECT phash, dhash, path, source FROM image_hashes").fetchall()
best, best_dist = None, None
for ph, dh, p, src in rows:
phd = hamming(ph, hx["phash"])
dhd = hamming(dh, hx["dhash"])
dist = min(phd, dhd)
if best_dist is None or dist < best_dist:
best, best_dist = (p, src), dist
if best_dist == 0:
break
if best and best_dist <= hamming_threshold:
return best, best_dist
return None, best_dist

View File

@@ -1,38 +1,69 @@
"""photo-pipeline: photo-ingest flow (v1). """photo-pipeline: photo-ingest flow (v2 — batched + checkpointed).
Stage 1 of the pipeline: hash incoming images, check against the persistent Hashes incoming images, checks against the persistent fingerprint DB
fingerprint DB (exact + near dupes), and register new ones. (exact + near dupes), and registers new ones.
Run via Prefect deployment on photo-pool (see prefect.yaml). Batching: processes in chunks of `batch_size` (default 1000), committing to
the DB after each chunk. Checkpointing is DB-native: files already in the DB
(by sha256) are skipped on resume — an interrupted run continues where it
stopped, never redoing work.
For very large trees (e.g. 58K files), scan_directory can be slow to walk;
use walk_files for a streaming generator when batch_size is set.
""" """
import sqlite3
from pathlib import Path from pathlib import Path
from prefect import flow, task from prefect import flow, task
import photo_db as db import photo_db as db
EXT_IMAGES = {".jpg", ".jpeg", ".png", ".heic", ".webp", ".gif", ".tif", ".tiff", ".bmp"}
@task
def scan_directory(base_dir: str) -> list[str]: def _is_image(p: Path) -> bool:
"""Enumerate image files in a directory tree.""" return p.is_file() and p.suffix.lower() in EXT_IMAGES
exts = {".jpg", ".jpeg", ".png", ".heic", ".webp", ".gif", ".tif", ".tiff", ".bmp"}
def walk_files(base_dir: str):
"""Stream image files under base_dir (generator — memory-safe for 50K+ files)."""
root = Path(base_dir) root = Path(base_dir)
if not root.exists(): if not root.exists():
raise FileNotFoundError(f"{root} does not exist") raise FileNotFoundError(f"{root} does not exist")
found = [ for p in root.rglob("*"):
str(p) if _is_image(p):
for p in root.rglob("*") yield str(p)
if p.is_file() and p.suffix.lower() in exts
]
print(f"Found {len(found)} images under {root}")
return found
@task @task
def check_duplicates(image_paths: list[str]) -> dict: def find_unprocessed(base_dir: str, batch_size: int, source: str = None) -> list[str]:
"""Check each image against the fingerprint DB. Returns classification.""" """Find the next batch of files NOT yet in the fingerprint DB."""
db.init_db()
conn = db.get_db()
batch = []
for p_str in walk_files(base_dir):
# skip if already registered for this source (or any source)
row = conn.execute(
"SELECT 1 FROM image_hashes WHERE sha256=?",
(db.sha256_file(p_str),),
).fetchone() if False else None
# cheap check: path already known?
known = conn.execute(
"SELECT 1 FROM image_hashes WHERE path=?", (p_str,)
).fetchone()
if known:
continue
batch.append(p_str)
if len(batch) >= batch_size:
break
conn.close()
print(f"find_unprocessed: {len(batch)} new files (batch_size={batch_size})")
return batch
@task
def check_and_register(image_paths: list[str], source: str) -> dict:
"""Hash + dedup-check + register a batch. Returns verdict counts."""
db.init_db() db.init_db()
conn = db.get_db() conn = db.get_db()
exact_dups = [] exact_dups = []
@@ -41,75 +72,85 @@ def check_duplicates(image_paths: list[str]) -> dict:
for p_str in image_paths: for p_str in image_paths:
p = Path(p_str) p = Path(p_str)
try: try:
match = db.check_exact_dup(conn, p) # exact dup by sha
sha = db.sha256_file(p)
match = conn.execute(
"SELECT path, source FROM image_hashes WHERE sha256=?", (sha,)
).fetchone()
if match: if match:
exact_dups.append((p_str, match)) exact_dups.append((p_str, match[0]))
continue continue
near, dist = db.check_near_dup(conn, p) # near dup by perceptual hash
if near: hx = db.hash_image(p)
near, dist = db._near_dup_lookup(conn, hx)
if near and dist <= 10:
near_dups.append((p_str, near, dist)) near_dups.append((p_str, near, dist))
continue continue
# new — register
with __import__("PIL.Image", fromlist=["Image"]).Image.open(p) as im:
w, h = im.size
conn.execute(
"INSERT INTO image_hashes (sha256, phash, dhash, file_size, width, height, path, source) "
"VALUES (?,?,?,?,?,?,?,?)",
(sha, hx["phash"], hx["dhash"], p.stat().st_size, w, h, str(p), source),
)
new_images.append(p_str) new_images.append(p_str)
except Exception as e: except Exception as e:
print(f" SKIP {p.name}: {e}") print(f" SKIP {p.name}: {type(e).__name__}: {e}")
conn.commit()
conn.close() conn.close()
result = { return {
"total": len(image_paths), "total": len(image_paths),
"exact_dups": len(exact_dups), "exact_dups": len(exact_dups),
"near_dups": len(near_dups), "near_dups": len(near_dups),
"new": len(new_images), "new": len(new_images),
"exact_dup_list": exact_dups[:50], "exact_dup_list": exact_dups[:20],
"near_dup_list": near_dups[:50], "near_dup_list": near_dups[:20],
"new_list": new_images, "new_list": new_images,
} }
print(
f"Check: {result['total']} total, "
f"{result['exact_dups']} exact dups, {result['near_dups']} near dups, "
f"{result['new']} new"
)
return result
@task
def register_new_images(image_paths: list[str], source: str) -> int:
"""Add hashes for confirmed-new images into the fingerprint DB."""
db.init_db()
conn = db.get_db()
registered = 0
for p_str in image_paths:
p = Path(p_str)
try:
if db.register_image(conn, p, source=source):
registered += 1
except Exception as e:
print(f" FAIL register {p.name}: {e}")
conn.commit()
conn.close()
print(f"Registered {registered} new images (source={source})")
return registered
@flow(name="photo-ingest") @flow(name="photo-ingest")
def photo_ingest(base_dir: str, source: str = "takeout", register: bool = True): def photo_ingest(base_dir: str, source: str = "takeout", batch_size: int = 1000,
"""Hash + dedup-check a folder against the persistent library.""" max_batches: int = None):
images = scan_directory(base_dir) """Hash + dedup-check a folder against the persistent library, in batches.
if not images:
print("No images found — nothing to do.")
return {"total": 0}
result = check_duplicates(images) Args:
base_dir: folder to scan
source: label for the batch (e.g. takeout, archive-pictures)
batch_size: files per batch/checkpoint (default 1000)
max_batches: stop after N batches (useful for testing) — None = all
"""
db.init_db()
processed_batches = 0
totals = {"exact_dups": 0, "near_dups": 0, "new": 0}
if register and result["new_list"]: while True:
n = register_new_images(result["new_list"], source=source) batch = find_unprocessed(base_dir, batch_size, source)
result["registered"] = n if not batch:
print("No more unprocessed files — done.")
break
result = check_and_register(batch, source)
totals["exact_dups"] += result["exact_dups"]
totals["near_dups"] += result["near_dups"]
totals["new"] += result["new"]
processed_batches += 1
print(f"batch {processed_batches} done: {result['total']} files, "
f"{result['new']} new, {result['exact_dups']} exact, {result['near_dups']} near")
if max_batches and processed_batches >= max_batches:
print(f"Stopped after {processed_batches} batches (max_batches={max_batches})")
break
return result totals["batches"] = processed_batches
return totals
if __name__ == "__main__": if __name__ == "__main__":
# Local run (no deployment)
import sys import sys
d = sys.argv[1] if len(sys.argv) > 1 else "/tmp/sample" d = sys.argv[1] if len(sys.argv) > 1 else "/tmp/sample"
r = photo_ingest(d) src = sys.argv[2] if len(sys.argv) > 2 else "cli-test"
bs = int(sys.argv[3]) if len(sys.argv) > 3 else 1000
mb = int(sys.argv[4]) if len(sys.argv) > 4 else None
r = photo_ingest(d, source=src, batch_size=bs, max_batches=mb)
print(r) print(r)

View File

@@ -29,14 +29,14 @@ deployments:
- name: ingest - name: ingest
version: null version: null
tags: [photo-pipeline] tags: [photo-pipeline]
description: "Hash + dedup-check a folder against the persistent fingerprint DB" description: "Hash + dedup-check a folder against the persistent fingerprint DB (batched, checkpointed)"
schedule: null schedule: null
flow_name: null flow_name: null
entrypoint: photo_ingest.py:photo_ingest entrypoint: photo_ingest.py:photo_ingest
parameters: parameters:
base_dir: /mnt/data/takeout base_dir: /mnt/data/takeout
source: takeout source: takeout
register: true batch_size: 1000
work_pool: work_pool:
name: photo-pool name: photo-pool
work_queue_name: null work_queue_name: null
@@ -68,6 +68,7 @@ deployments:
parameters: parameters:
base_dir: /mnt/data/takeout base_dir: /mnt/data/takeout
move: false move: false
max_files: 0
work_pool: work_pool:
name: photo-pool name: photo-pool
work_queue_name: null work_queue_name: null

View File

@@ -86,9 +86,37 @@ def _move(p: Path, dest_root: Path, src_root: Path):
shutil.move(str(p), str(dest)) shutil.move(str(p), str(dest))
def _slice_dir(base_dir: str, max_files: int) -> str:
"""Copy first N images into a temp dir for CleanVision to audit."""
import shutil
import tempfile
from pathlib import Path
src = Path(base_dir)
tmp = Path(tempfile.mkdtemp(prefix="cvslice_"))
exts = {".jpg", ".jpeg", ".png", ".webp", ".gif", ".heic", ".tif", ".bmp"}
n = 0
for p in src.rglob("*"):
if p.is_file() and p.suffix.lower() in exts:
shutil.copy2(p, tmp / p.name)
n += 1
if n >= max_files:
break
print(f"_slice_dir: copied {n} files to {tmp}")
return str(tmp)
@flow(name="photo-quality-scan") @flow(name="photo-quality-scan")
def quality_scan(base_dir: str, move: bool = False, notify: bool = True): def quality_scan(base_dir: str, move: bool = False, notify: bool = True, max_files: int = None):
"""Audit image quality with CleanVision; classify into keep/review/delete.""" """Audit image quality with CleanVision; classify into keep/review/delete.
max_files: if set, only audit the first N image files (slices huge folders
into reviewable chunks — prevents OOM on 50K-file trees).
"""
if max_files: # 0/None = unlimited
base_dir = _slice_dir(base_dir, max_files)
audit = audit_folder(base_dir) audit = audit_folder(base_dir)
print(f"Issue summary: {audit['summary']}") print(f"Issue summary: {audit['summary']}")
result = classify_and_sort(base_dir, audit["per_image"], move=move) result = classify_and_sort(base_dir, audit["per_image"], move=move)