181 lines
5.8 KiB
Python
181 lines
5.8 KiB
Python
"""photo-pipeline: persistent SQLite fingerprint + Takeout tracking DB.
|
|
|
|
Two responsibilities:
|
|
1. image_hashes — permanent library identity (sha256 exact + phash/dhash near)
|
|
2. takeout_archives — shipment tracking for Takeout dumps (the "lost track" problem)
|
|
3. batches — which source batch fed which registration
|
|
|
|
Hashing is the foundation; this DB is where the hashes live and are queried.
|
|
Immich keeps its own hashes in Postgres as the in-library layer.
|
|
"""
|
|
|
|
import hashlib
|
|
import sqlite3
|
|
from pathlib import Path
|
|
|
|
from imagehash import dhash, phash
|
|
from PIL import Image
|
|
|
|
DB_PATH = Path(__file__).parent / "photo_pipeline.db"
|
|
|
|
SCHEMA = """
|
|
CREATE TABLE IF NOT EXISTS image_hashes (
|
|
sha256 TEXT PRIMARY KEY,
|
|
phash TEXT NOT NULL,
|
|
dhash TEXT NOT NULL,
|
|
file_size INTEGER,
|
|
width INTEGER,
|
|
height INTEGER,
|
|
exif_date TEXT,
|
|
path TEXT,
|
|
source TEXT, -- takeout | archive | phone | immich
|
|
in_immich INTEGER DEFAULT 0,
|
|
added_at TEXT DEFAULT (datetime('now'))
|
|
);
|
|
|
|
CREATE TABLE IF NOT EXISTS takeout_archives (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
export_id TEXT,
|
|
archive_name TEXT UNIQUE,
|
|
sha256 TEXT, -- hash of the ARCHIVE file itself
|
|
size_bytes INTEGER,
|
|
created_at TEXT,
|
|
downloaded_at TEXT,
|
|
extracted_at TEXT,
|
|
extract_path TEXT,
|
|
file_count INTEGER,
|
|
status TEXT DEFAULT 'pending' -- pending|downloading|downloaded|extracted|registered
|
|
);
|
|
|
|
CREATE TABLE IF NOT EXISTS batches (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
name TEXT,
|
|
archive_id INTEGER REFERENCES takeout_archives(id),
|
|
source TEXT,
|
|
started_at TEXT,
|
|
finished_at TEXT,
|
|
status TEXT
|
|
);
|
|
|
|
CREATE INDEX IF NOT EXISTS idx_image_hashes_phash ON image_hashes(phash);
|
|
CREATE INDEX IF NOT EXISTS idx_image_hashes_dhash ON image_hashes(dhash);
|
|
"""
|
|
|
|
|
|
def get_db():
|
|
conn = sqlite3.connect(DB_PATH)
|
|
conn.execute("PRAGMA journal_mode=WAL")
|
|
conn.execute("PRAGMA foreign_keys=ON")
|
|
return conn
|
|
|
|
|
|
def init_db():
|
|
with get_db() as conn:
|
|
conn.executescript(SCHEMA)
|
|
|
|
|
|
def sha256_file(path: Path) -> str:
|
|
h = hashlib.sha256()
|
|
with open(path, "rb") as f:
|
|
for chunk in iter(lambda: f.read(1024 * 1024), b""):
|
|
h.update(chunk)
|
|
return h.hexdigest()
|
|
|
|
|
|
def hash_image(path: Path, hash_size: int = 8):
|
|
"""Perceptual hashes for one image file."""
|
|
with Image.open(path) as im:
|
|
im = im.convert("RGB")
|
|
return {
|
|
"phash": str(phash(im, hash_size=hash_size)),
|
|
"dhash": str(dhash(im, hash_size=hash_size)),
|
|
}
|
|
|
|
|
|
def register_image(conn, path: Path, source: str, exif_date: str | None = None):
|
|
"""Insert one image's hashes if not already present. Returns True if new."""
|
|
sha = sha256_file(path)
|
|
existing = conn.execute("SELECT 1 FROM image_hashes WHERE sha256=?", (sha,)).fetchone()
|
|
if existing:
|
|
return False
|
|
with Image.open(path) as im:
|
|
w, h = im.size
|
|
hx = hash_image(path)
|
|
conn.execute(
|
|
"INSERT INTO image_hashes (sha256, phash, dhash, file_size, width, height, exif_date, path, source) "
|
|
"VALUES (?,?,?,?,?,?,?,?,?)",
|
|
(sha, hx["phash"], hx["dhash"], path.stat().st_size, w, h, exif_date, str(path), source),
|
|
)
|
|
return True
|
|
|
|
|
|
def check_exact_dup(conn, path: Path):
|
|
"""Return matching library path if this file's sha256 already exists."""
|
|
sha = sha256_file(path)
|
|
row = conn.execute("SELECT path, source FROM image_hashes WHERE sha256=?", (sha,)).fetchone()
|
|
return row
|
|
|
|
|
|
def check_near_dup(conn, path: Path, hamming_threshold: int = 10):
|
|
"""Return nearest-matching library path if phash+dhash are close enough."""
|
|
hx = hash_image(path)
|
|
rows = conn.execute("SELECT phash, dhash, path, source FROM image_hashes").fetchall()
|
|
best, best_dist = None, None
|
|
for ph, dh, p, src in rows:
|
|
phd = hamming(ph, hx["phash"])
|
|
dhd = hamming(dh, hx["dhash"])
|
|
dist = min(phd, dhd)
|
|
if best_dist is None or dist < best_dist:
|
|
best, best_dist = (p, src), dist
|
|
if best_dist == 0:
|
|
break
|
|
if best and best_dist <= hamming_threshold:
|
|
return best, best_dist
|
|
return None, best_dist
|
|
|
|
|
|
def hamming(a: str, b: str) -> int:
|
|
return sum(1 for x, y in zip(a, b) if x != y)
|
|
|
|
|
|
def register_takeout_archive(conn, archive_name: str, path: Path) -> int:
|
|
"""Record a downloaded archive (or detect it's already known). Returns row id."""
|
|
sha = sha256_file(path)
|
|
row = conn.execute(
|
|
"SELECT id FROM takeout_archives WHERE archive_name=?", (archive_name,)
|
|
).fetchone()
|
|
if row:
|
|
conn.execute(
|
|
"UPDATE takeout_archives SET sha256=?, size_bytes=?, downloaded_at=datetime('now'), status='downloaded' WHERE id=?",
|
|
(sha, path.stat().st_size, row[0]),
|
|
)
|
|
return row[0]
|
|
cur = conn.execute(
|
|
"INSERT INTO takeout_archives (archive_name, sha256, size_bytes, downloaded_at, status) "
|
|
"VALUES (?,?,?,datetime('now'),'downloaded')",
|
|
(archive_name, sha, path.stat().st_size),
|
|
)
|
|
return cur.lastrowid
|
|
|
|
|
|
if __name__ == "__main__":
|
|
init_db()
|
|
print(f"DB ready at {DB_PATH}")
|
|
|
|
|
|
def _near_dup_lookup(conn, hx: dict, hamming_threshold: int = 10):
|
|
"""Find nearest perceptual-hash match, given precomputed hashes (avoids re-hash)."""
|
|
rows = conn.execute("SELECT phash, dhash, path, source FROM image_hashes").fetchall()
|
|
best, best_dist = None, None
|
|
for ph, dh, p, src in rows:
|
|
phd = hamming(ph, hx["phash"])
|
|
dhd = hamming(dh, hx["dhash"])
|
|
dist = min(phd, dhd)
|
|
if best_dist is None or dist < best_dist:
|
|
best, best_dist = (p, src), dist
|
|
if best_dist == 0:
|
|
break
|
|
if best and best_dist <= hamming_threshold:
|
|
return best, best_dist
|
|
return None, best_dist
|