US07-03: Harden Media and Metadata Edge Cases (#85)

This commit was merged in pull request #85.
This commit is contained in:
2026-08-17 01:12:57 +02:00
parent d2e44b5657
commit 3fa35fe21e
18 changed files with 1525 additions and 53 deletions

View File

@@ -56,6 +56,30 @@ def read_keyword_sets(paths: Iterable[str]) -> dict[str, set[str]]:
return out
def read_all(path: str) -> dict | None:
"""Every tag exiftool can read from ``path``, or ``None`` when it cannot answer.
This is the snapshot an EXIF checkpoint compares against: proving that a write
preserved the fields it does not own requires knowing all of them, not just the
ones being written (US07-03). ``None`` (exiftool missing, unreadable file,
unparsable output) is not an empty snapshot — a caller must not read it as
"nothing was there".
"""
try:
result = subprocess.run(
["exiftool", "-m", "-j", "-G0:1", path], capture_output=True, text=True
)
except FileNotFoundError:
return None
try:
records = json.loads(result.stdout or "[]")
except ValueError:
return None
if not records:
return None
return {k: v for k, v in records[0].items() if k != "SourceFile"}
def apply_keywords(path: str, *, add: Iterable[str] = (), remove: Iterable[str] = ()) -> bool:
"""Idempotently add/remove keywords in Keywords + Subject; preserve all else."""
args = ["exiftool", "-m", "-overwrite_original"]

View File

@@ -15,6 +15,8 @@ from __future__ import annotations
from pathlib import Path
from photo_pipeline import imaging
MODEL_ID = "AdamCodd/vit-base-nsfw-detector"
BATCH = 16
@@ -54,13 +56,17 @@ class NsfwModel:
self._ensure_loaded()
import numpy as np
import torch
from PIL import Image, ImageFile
from PIL import Image
ImageFile.LOAD_TRUNCATED_IMAGES = True
def preprocess(image):
image = image.convert("RGB").resize((self._size, self._size), Image.BILINEAR)
array = (np.asarray(image, dtype="float32") / 255.0 - 0.5) / 0.5
# The donor set ``ImageFile.LOAD_TRUNCATED_IMAGES = True`` here. That flag is
# process-global: in this application the same process also hashes files and
# renders previews, and those must keep failing loudly on a truncated file
# rather than quietly working on half of one (US07-03). An unreadable image
# is skipped instead — it stays unscored, and therefore visibly undecided.
def preprocess(path):
with imaging.open_image(path) as image:
small = image.convert("RGB").resize((self._size, self._size), Image.BILINEAR)
array = (np.asarray(small, dtype="float32") / 255.0 - 0.5) / 0.5
return torch.from_numpy(array).permute(2, 0, 1)
results: list[tuple[str, float]] = []
@@ -69,9 +75,10 @@ class NsfwModel:
tensors, batch_paths = [], []
for path in items[start : start + self.batch]:
try:
tensors.append(preprocess(Image.open(path)))
tensors.append(preprocess(path))
batch_paths.append(path)
except Exception:
except (imaging.MediaError, OSError, ValueError):
# One bad file must not cost the batch its other fifteen.
continue
if not tensors:
continue