Files
photoanalyzer/tests/e2e/test_crash_recovery.py

275 lines
10 KiB
Python

"""Process death at the newer control points (US07-04, concept §18).
The rename and archive journals already prove crash safety at each of their
transitions (tests/integration/test_rename_recovery.py,
tests/integration/test_archive_recovery.py). The three transitions covered here
are the remaining ones where a kill leaves the world and the database disagreeing:
- ``exif:written`` — keywords are on disk, nothing about them is recorded;
- ``upload:accepted``— the uploader finished, no outcome is stored;
- ``job:item_done`` — one item is durably done, the job is not finished.
Each test kills a real child process at the barrier and then asserts what a
restart does: resume idempotently, or say plainly that a human has to look. Never
"assume it worked".
"""
from __future__ import annotations
import os
import shutil
import subprocess
import sys
import uuid
from datetime import datetime, timezone
from pathlib import Path
import numpy as np
import pytest
from PIL import Image
from sqlalchemy import select
from photo_pipeline.config import Config
from photo_pipeline.db import create_db_engine, create_session_factory, run_migrations
from photo_pipeline.faults import EXIF_WRITTEN, JOB_ITEM_DONE, UPLOAD_ACCEPTED
from photo_pipeline.integrations import exiftool
from photo_pipeline.jobs.worker import Worker
from photo_pipeline.models import Asset, SafetyReview
from photo_pipeline.services import exif_checkpoint, hashing
from photo_pipeline.services.jobs import ItemState, JobService, JobState
from photo_pipeline.services.safety import SafetyService
from photo_pipeline.services.upload_batches import BatchState, UploadBatchService
from photo_pipeline.services.upload_verification import retry_blockers
from photo_pipeline.services.uploads import UploadService
from tests.e2e._pipeline_harness import (
SILENT_UPLOADER,
FakeImmich,
fake_uploader,
mark_upload_ready,
seed_album,
)
REPO = Path(__file__).resolve().parents[2]
NOW = datetime(2026, 1, 1, tzinfo=timezone.utc)
def _child(script: str, *args: str, barrier: str, tmp_path: Path) -> None:
"""Run ``script`` in a child that dies at ``barrier``; assert it really died."""
path = tmp_path / f"child_{barrier.replace(':', '_')}.py"
path.write_text(script.format(repo=str(REPO)))
env = dict(os.environ)
env["PHOTO_PIPELINE_FAULT_AFTER"] = barrier
result = subprocess.run(
[sys.executable, str(path), *args], env=env, capture_output=True
)
assert result.returncode in (9, -9), (
f"child should have been killed at {barrier}, got {result.returncode}: "
f"{result.stderr.decode(errors='replace')[-400:]}"
)
def _env(tmp_path, **extra):
(tmp_path / "data").mkdir(exist_ok=True)
lib = tmp_path / "lib"
lib.mkdir(exist_ok=True)
config = Config.from_env(
{
"PHOTO_PIPELINE_DATA_DIR": str(tmp_path / "data"),
"PHOTO_PIPELINE_LIBRARY_ROOTS": str(lib),
**extra,
}
)
run_migrations(config.database_url)
return config, create_session_factory(create_db_engine(config.database_url)), lib
def _image(path: Path, seed: int = 3) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
pixels = np.random.default_rng(seed).integers(0, 256, (64, 96, 3), dtype=np.uint8)
Image.fromarray(pixels).save(path, quality=90)
def _register(sf, path: Path) -> str:
asset_id = str(uuid.uuid4())
with sf() as session:
session.add(
Asset(
id=asset_id,
original_path=str(path),
current_path=str(path),
discovered_at=NOW,
hash_version=1,
byte_size=path.stat().st_size,
current_sha256=hashing.sha256_file(path),
)
)
session.commit()
return asset_id
# ── EXIF written, checkpoint not recorded ────────────────────────────────────
DECIDE_SCRIPT = """
import sys
sys.path.insert(0, {repo!r})
from photo_pipeline.db import create_db_engine, create_session_factory
from photo_pipeline.services.safety import SafetyService
db_url, asset_id = sys.argv[1], sys.argv[2]
sf = create_session_factory(create_db_engine(db_url))
SafetyService(sf).decide(asset_id, "nsfw")
"""
@pytest.mark.skipif(shutil.which("exiftool") is None, reason="exiftool not installed")
def test_a_crash_after_the_exif_write_leaves_nothing_verified_and_re_runs_cleanly(tmp_path):
config, sf, lib = _env(tmp_path)
path = lib / "a.jpg"
_image(path)
asset_id = _register(sf, path)
original_sha = hashing.sha256_file(path)
_child(
DECIDE_SCRIPT, config.database_url, asset_id, barrier=EXIF_WRITTEN, tmp_path=tmp_path
)
# The file changed, but the application claims nothing about it: no decision,
# no projection, and the stored hash is still the pre-write one.
assert "nsfw" in exiftool.read_keyword_sets([str(path)])[str(path)]
assert hashing.sha256_file(path) != original_sha
with sf() as session:
assert session.scalars(select(SafetyReview)).all() == []
assert session.get(Asset, asset_id).current_sha256 == original_sha
assert exif_checkpoint.state_for(sf, asset_id, "safety") is None
# Re-running is the recovery: the write is idempotent, so the second attempt
# verifies and records what the first one only did to the file.
review = SafetyService(sf).decide(asset_id, "nsfw")
assert review["exif_verified"] is True
assert exif_checkpoint.state_for(sf, asset_id, "safety") == exif_checkpoint.VERIFIED
keywords = exiftool.read_keyword_sets([str(path)])[str(path)]
assert "nsfw" in keywords and "sfw" not in keywords
with sf() as session:
asset = session.get(Asset, asset_id)
assert asset.current_sha256 == hashing.sha256_file(path)
# ── uploader accepted, outcome not persisted ─────────────────────────────────
UPLOAD_SCRIPT = """
import sys
sys.path.insert(0, {repo!r})
from photo_pipeline.config import Config
from photo_pipeline.db import create_db_engine, create_session_factory
from photo_pipeline.services.upload_batches import UploadBatchService
db_url, data_dir, lib, binary, server, batch_id = sys.argv[1:7]
config = Config.from_env(
{{
"PHOTO_PIPELINE_DATA_DIR": data_dir,
"PHOTO_PIPELINE_LIBRARY_ROOTS": lib,
"PHOTO_PIPELINE_IMMICH_GO_BINARY": binary,
"PHOTO_PIPELINE_IMMICH_SERVER_URL": server,
"PHOTO_PIPELINE_IMMICH_API_KEY": "sentinel",
}}
)
sf = create_session_factory(create_db_engine(db_url))
UploadBatchService(sf, config=config).run(batch_id)
"""
def test_a_crash_after_the_uploader_accepted_requires_verification(tmp_path):
"""immich-go exited cleanly and the server may hold every file, but nothing was
written down. Recovery must not guess success — and must not blindly retry."""
seeded = seed_album(tmp_path)
mark_upload_ready(seeded)
immich = FakeImmich()
binary = fake_uploader(tmp_path, SILENT_UPLOADER)
config = Config.from_env(
{
"PHOTO_PIPELINE_DATA_DIR": str(seeded.data),
"PHOTO_PIPELINE_LIBRARY_ROOTS": str(seeded.lib),
"PHOTO_PIPELINE_IMMICH_GO_BINARY": str(binary),
"PHOTO_PIPELINE_IMMICH_SERVER_URL": immich.url,
"PHOTO_PIPELINE_IMMICH_API_KEY": "sentinel",
}
)
engine = create_db_engine(config.database_url)
sf = create_session_factory(engine)
service = UploadBatchService(sf, config=config)
report = UploadService(sf, config=config).preflight(["rome"])
assert report["state"] == "ready", report["blockers"]
batch_id = service.create(["rome"], token=report["token"])[0]["id"]
try:
# The batch is claimed by the child, which dies once the uploader has run.
_child(
UPLOAD_SCRIPT,
config.database_url,
str(seeded.data),
str(seeded.lib),
str(binary),
immich.url,
batch_id,
barrier=UPLOAD_ACCEPTED,
tmp_path=tmp_path,
)
finally:
immich.stop()
assert service.get(batch_id)["state"] == BatchState.RUNNING # lane still held
recovered = service.recover()
assert recovered == {"interrupted": 1}
batch = service.get(batch_id)
assert batch["state"] == BatchState.UNKNOWN and batch["error_code"] == "interrupted"
assert [b["code"] for b in retry_blockers(batch)] == ["requires_verification"]
assert all(item["state"] == "pending" for item in batch["items"]), "nothing claimed as sent"
# ── one item done, the job unfinished ────────────────────────────────────────
WORKER_SCRIPT = """
import sys
sys.path.insert(0, {repo!r})
from pathlib import Path
from photo_pipeline.db import create_db_engine, create_session_factory
from photo_pipeline.jobs.worker import Worker
db_url, log = sys.argv[1], Path(sys.argv[2])
def handler(item_key, ctx):
with log.open("a") as handle:
handle.write(item_key + "\\n")
sf = create_session_factory(create_db_engine(db_url))
Worker(sf, {{"scan": handler}}, "killable", lease_seconds=1).run_once()
"""
def test_a_crash_between_items_resumes_without_running_the_done_item_twice(tmp_path):
config, sf, lib = _env(tmp_path)
service = JobService(sf)
job = service.enqueue("scan", items=["a", "b", "c"])
log = tmp_path / "handled.log"
_child(
WORKER_SCRIPT, config.database_url, str(log), barrier=JOB_ITEM_DONE, tmp_path=tmp_path
)
crashed = log.read_text().split()
assert crashed == ["a"], "the child should have died right after its first item"
assert service.get(job["id"])["state"] == JobState.RUNNING
# A fresh worker takes over once the dead lease expires.
import time
time.sleep(1.1) # the child's lease is one second long
fresh = Worker(sf, {"scan": lambda item, ctx: log.open("a").write(item + "\n")}, "alive")
fresh.run_once()
assert service.get(job["id"])["state"] == JobState.SUCCEEDED
handled = log.read_text().split()
assert sorted(handled) == ["a", "b", "c"], f"an item ran twice or not at all: {handled}"
assert service.progress(job["id"])["by_state"] == {ItemState.SUCCEEDED: 3}