Compare commits
1 Commits
chore/E09-
...
us/US07-01
| Author | SHA1 | Date | |
|---|---|---|---|
| dc1a64203b |
@@ -80,7 +80,8 @@ are the authoritative backlog.
|
|||||||
uncertain. Work only on the claimed story and its generated feature branch.
|
uncertain. Work only on the claimed story and its generated feature branch.
|
||||||
4. Read the entire issue, linked specification, dependencies, and acceptance criteria.
|
4. Read the entire issue, linked specification, dependencies, and acceptance criteria.
|
||||||
Reconcile them with the concept before designing or changing code.
|
Reconcile them with the concept before designing or changing code.
|
||||||
5. Inspect the legacy CLI donors before replacing applicable behavior. Update the donor
|
5. Inspect the legacy CLI donors — frozen in `legacy_cli_archive/` since US07-01,
|
||||||
|
with their ledger — before replacing applicable behavior. Update the donor
|
||||||
ledger and characterization tests required by the story.
|
ledger and characterization tests required by the story.
|
||||||
6. Implement every acceptance criterion and its automated tests.
|
6. Implement every acceptance criterion and its automated tests.
|
||||||
7. Run story-specific tests and the accumulated regression suite required by the epic.
|
7. Run story-specific tests and the accumulated regression suite required by the epic.
|
||||||
|
|||||||
33
README.md
33
README.md
@@ -208,3 +208,36 @@ work_item/scripts/python -m pytest -m phase_f -q
|
|||||||
offline browsing, restore, collision, keyboard confirmation, and reload).
|
offline browsing, restore, collision, keyboard confirmation, and reload).
|
||||||
|
|
||||||
Phases A–E remain green in the full run above.
|
Phases A–E remain green in the full run above.
|
||||||
|
|
||||||
|
## Legacy CLI archive
|
||||||
|
|
||||||
|
The command-line tools this application was extracted from are frozen in
|
||||||
|
`legacy_cli_archive/` (US07-01): the original sources, their docs, the dependency
|
||||||
|
lock they were last verified against, schema notes, a redacted sample
|
||||||
|
configuration, the donor ledger, and a checksum for every file.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd legacy_cli_archive && shasum -a 256 -c CHECKSUMS.sha256 # verify the archive
|
||||||
|
work_item/scripts/python -m pytest tests/unit/test_legacy_archive.py -q # lint it
|
||||||
|
```
|
||||||
|
|
||||||
|
They are reference material and rollback evidence only. No module under
|
||||||
|
`photo_pipeline/` imports or executes them, the archive is not on the application's
|
||||||
|
import path, and `tests/unit/test_legacy_archive.py` enforces that along with the
|
||||||
|
checksums and the redaction. Only the two suites that compare *against* the donors —
|
||||||
|
`tests/characterization/` and `tests/integration/test_safety_parity.py` — put the
|
||||||
|
archived sources on `sys.path`.
|
||||||
|
|
||||||
|
The last path-keyed state they owned, `nsfw_scores.csv`, is imported once and then
|
||||||
|
left alone:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
work_item/scripts/python -m photo_pipeline import-legacy-scores /path/to/nsfw_scores.csv --dry-run
|
||||||
|
```
|
||||||
|
|
||||||
|
The import writes scored-but-unreviewed `safety_reviews` rows onto stable asset ids,
|
||||||
|
never invents an asset for an unknown path, never overwrites a human decision, and
|
||||||
|
writes a reconciliation report to the data directory saying exactly what it did.
|
||||||
|
`legacy_cli_archive/donor_ledger.yaml` records every migrated behavior with its
|
||||||
|
target, the tests that pin the donor, the tests that prove the replacement, and each
|
||||||
|
intentional delta; rows still marked `pending` name the story that will resolve them.
|
||||||
|
|||||||
22
legacy_cli_archive/CHECKSUMS.sha256
Normal file
22
legacy_cli_archive/CHECKSUMS.sha256
Normal file
@@ -0,0 +1,22 @@
|
|||||||
|
d098bbde13d2ebc872ca781e244cc1e48b97cc551fedaca3251d6a5278a15234 src/compare_models.py
|
||||||
|
7d0f68cb95fbd6603e4c558620098b46929ebdb6fd91a598bbf84f26eea35e6c src/nsfw_tag.py
|
||||||
|
35329c53570e215cef15f59429c1a5251a0448b90522b0910ab9d68efe1bc307 src/nsfwtag/__init__.py
|
||||||
|
7f52b55e4f3b77eda3657d7cd2272c7cca8422100e241671de495603c1126ee2 src/nsfwtag/__main__.py
|
||||||
|
6f5a97114e0d87d272ce22d065dc31fa0dd72ff8590d11ad14cb1c1f486d339c src/nsfwtag/bench.py
|
||||||
|
29498fac1d73ba2b7420ffe1ffe49a87f684ec299e2c3b9c6e14e63e95643ba4 src/nsfwtag/exif.py
|
||||||
|
934e82c402813ebf503e5a20eb03d5103f84df13117bc4dd1fd95deaa01ab263 src/nsfwtag/README.md
|
||||||
|
68125e6184c7e4d2a5b0153f2155675753ab0b1329d1dbc4933d4769492ec4be src/nsfwtag/review.html
|
||||||
|
a597eab74803dd31452b5d70abf7d6d6320eb17a61596c891b009da15d624e98 src/nsfwtag/scoring.py
|
||||||
|
674969a18e58ee511a2abf574a875db92f2f524769b1b614d0d3662ab879a709 src/nsfwtag/server.py
|
||||||
|
24a7c8d029da6e97d46b110a9fe9dbb3900127f8ae3c142f634fdd91d6c45243 src/nsfwtag/webapp.py
|
||||||
|
2c2ea558f1b9095c1150f078ea29c8e0f180cc045a9d1097b83f61bfb145295b src/photo_analyzer.py
|
||||||
|
5fb7ce2f977a17da1f501c45d7985328318ae14298a3f347eab46a6e3f02aab3 src/test_dedup.py
|
||||||
|
1673dc76cc60aba56a5b065d9bc9f5dda00342cf2a79d000a3f74686fb7cffec src/test_nsfw_skip.py
|
||||||
|
55aa82f348e0e90be0163ba6aa278e5fc8e445996e5335774917e5ef59ad3563 src/webapp/__init__.py
|
||||||
|
ba4342bd0175591a2121f063f14a78a602d5679bcfcf24aa14c8d4b8cd5da1a2 src/webapp/__main__.py
|
||||||
|
a60f24035a989909467778f888d62854c1afc2702f47b8d3cca32bf3cf04d5cf src/webapp/analyzer.html
|
||||||
|
018bcc2f770d716444b456db58d6b4a41800138a98598b3c27ba503ab22d939e src/webapp/page.py
|
||||||
|
5f599b107b2b117ca118b6fbdec5e5786a9ab4eea424aa9af1bfea7bf87cebab src/webapp/query.py
|
||||||
|
875cce697caa02717c85a2707ba935c46dd9ace18428fd4abb13105b7a8e24d9 src/webapp/README.md
|
||||||
|
af7d0d72d245b4bbb1e0e30f9708239697543212d4171defd1bfe26dd24a4f05 src/webapp/runner.py
|
||||||
|
cddf0555b06fe7e4daa14309baaaf91130f8506922cb0ffb6507d76cbe39add4 src/webapp/server.py
|
||||||
80
legacy_cli_archive/README.md
Normal file
80
legacy_cli_archive/README.md
Normal file
@@ -0,0 +1,80 @@
|
|||||||
|
# Legacy CLI archive (US07-01)
|
||||||
|
|
||||||
|
Frozen, read-only sources of the command-line tools this application was extracted
|
||||||
|
from. They are **reference material and rollback evidence** — provenance for
|
||||||
|
behavior that now lives in `photo_pipeline/`, and the only way to answer "what did
|
||||||
|
the original actually do?" once the replacement has drifted.
|
||||||
|
|
||||||
|
> **Nothing here is production code.** No module under `photo_pipeline/` imports or
|
||||||
|
> executes anything in this directory, and this directory is not on the application's
|
||||||
|
> import path. `tests/unit/test_legacy_archive.py` enforces both, along with the
|
||||||
|
> checksums and the redaction below.
|
||||||
|
|
||||||
|
## What is here
|
||||||
|
|
||||||
|
| Path | Role |
|
||||||
|
|---|---|
|
||||||
|
| `src/photo_analyzer.py` | the analysis CLI: discovery, hashing/dedup, vision analysis, EXIF writing, SQLite schema, album naming |
|
||||||
|
| `src/nsfwtag/` | NSFW scoring, EXIF safety keywords, and the review server (`__init__` 1.1.0) |
|
||||||
|
| `src/webapp/` | the stdlib review web app: FTS search, stats, subprocess runner, HTML shell (`__init__` 0.1.0) |
|
||||||
|
| `src/nsfw_tag.py` | thin backwards-compatible entry point for `nsfwtag` |
|
||||||
|
| `src/compare_models.py` | dev-only model comparison script |
|
||||||
|
| `src/test_dedup.py`, `src/test_nsfw_skip.py` | the CLIs' own standalone self-checks (never pytest suites) |
|
||||||
|
| `donor_ledger.yaml` | the donor ledger: every migrated behavior, its target, its tests, and every intentional delta |
|
||||||
|
| `requirements-lock.txt` | the dependency versions the frozen sources were last verified against |
|
||||||
|
| `photo_analyzer.env.sample` | the CLI's configuration surface, with every value replaced by a placeholder |
|
||||||
|
| `CHECKSUMS.sha256` | SHA-256 of every archived source file |
|
||||||
|
|
||||||
|
`photo_analyzer.py` carries no `__version__`; its identity is its checksum, recorded
|
||||||
|
in `CHECKSUMS.sha256` and taken at commit `9b7ee6b` (the merge of US06-06, the last
|
||||||
|
commit before archival).
|
||||||
|
|
||||||
|
## Verifying the archive
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd legacy_cli_archive && shasum -a 256 -c CHECKSUMS.sha256
|
||||||
|
```
|
||||||
|
|
||||||
|
Any edit to an archived source must be accompanied by a regenerated checksum file
|
||||||
|
and a note here explaining why a *frozen* archive changed — the normal answer being
|
||||||
|
that it should not.
|
||||||
|
|
||||||
|
## Schema notes
|
||||||
|
|
||||||
|
`photo_analyzer.py` owned a path-keyed SQLite database (`SCHEMA`, near the top of the
|
||||||
|
file):
|
||||||
|
|
||||||
|
- `photos(id, path UNIQUE, status, phash, file_sha1, dup_of, description, tags,
|
||||||
|
people_count, setting, time_of_day, season, mood, location_hint, approx_year,
|
||||||
|
raw_response, error_message, analyzed_at, exif_written_at)`;
|
||||||
|
- `photos_fts` — an FTS5 external-content index over `path, description, tags, mood,
|
||||||
|
location_hint`, kept in sync by insert/update/delete triggers;
|
||||||
|
- late columns (`phash`, `file_sha1`, `dup_of`) were added by an in-code
|
||||||
|
`_migrate_schema()` rather than a migration tool, and their indexes are created
|
||||||
|
only after the `ALTER`.
|
||||||
|
|
||||||
|
The replacement keeps the same analysis fields but re-keys everything to a stable
|
||||||
|
`assets.id` (Alembic migrations `0001`…), because a path is not an identity: the
|
||||||
|
donor's `path UNIQUE` is exactly what broke on every move and rename.
|
||||||
|
|
||||||
|
`nsfwtag` kept its safety scores outside the database in `nsfw_scores.csv`
|
||||||
|
(`path,nsfw_score`, four decimals, unreadable rows dropped). That file is no longer a
|
||||||
|
source of truth; `photo_pipeline/services/legacy_import.py` imports it into
|
||||||
|
`assets.safety_score` and reports exactly what matched, what did not, and why.
|
||||||
|
|
||||||
|
## Redaction
|
||||||
|
|
||||||
|
The archive contains no credentials. `photo_analyzer.env.sample` documents the
|
||||||
|
configuration surface (`LLM_API_KEY`, `LLM_BASE_URL`, `LLM_MODEL`, and the tuning
|
||||||
|
variables) with placeholder values only; the CLI itself never contained a key, it
|
||||||
|
read one from `photo_analyzer.env` or the environment. No `.env`, database, log, CSV,
|
||||||
|
or photo from the author's library is archived.
|
||||||
|
|
||||||
|
## Why these tools were retired
|
||||||
|
|
||||||
|
Each behavior's fate is recorded per row in `donor_ledger.yaml`: `reuse`, `extract`,
|
||||||
|
`refactor`, or `replace`, with the target module, the characterization tests that
|
||||||
|
pinned the donor's behavior, the parity tests the replacement passes, and — where the
|
||||||
|
replacement deliberately does something else or nothing at all — a `delta` saying so.
|
||||||
|
Rows still marked `pending` name the backlog story that will resolve them; they are
|
||||||
|
the honest list of what has *not* been carried over yet.
|
||||||
@@ -12,7 +12,15 @@
|
|||||||
# (if anything) carries over
|
# (if anything) carries over
|
||||||
# Every row needs either `tests` (existing test IDs, module::function) or
|
# Every row needs either `tests` (existing test IDs, module::function) or
|
||||||
# `pending_story` (the backlog story that will characterize/deliver it).
|
# `pending_story` (the backlog story that will characterize/deliver it).
|
||||||
# status: characterized | pending
|
# status: characterized — donor behavior pinned by characterization tests
|
||||||
|
# resolved — replacement shipped; `parity` names the tests that prove
|
||||||
|
# it, and `delta` states every intentional difference
|
||||||
|
# pending — not migrated yet; `pending_story` says which story will
|
||||||
|
# parity: test ids (path::function) in any suite, proving the replacement
|
||||||
|
# delta: what the replacement deliberately does differently, or not at all
|
||||||
|
#
|
||||||
|
# Archived by US07-01: the sources referenced below now live beside this file in
|
||||||
|
# src/ and are frozen (see README.md). Nothing in photo_pipeline imports them.
|
||||||
|
|
||||||
rows:
|
rows:
|
||||||
# ── photo_analyzer.py ──────────────────────────────────────────────────────
|
# ── photo_analyzer.py ──────────────────────────────────────────────────────
|
||||||
@@ -80,8 +88,12 @@ rows:
|
|||||||
Copying a primary's analysis into variant rows survives, but keyed by
|
Copying a primary's analysis into variant rows survives, but keyed by
|
||||||
asset_id and recorded as stage state instead of raw row copies.
|
asset_id and recorded as stage state instead of raw row copies.
|
||||||
target: photo_pipeline/services/duplicates.py
|
target: photo_pipeline/services/duplicates.py
|
||||||
pending_story: US01-04
|
parity:
|
||||||
status: pending
|
- tests/integration/test_duplicate_engine.py::test_exact_copies_form_auto_decided_cluster
|
||||||
|
- tests/integration/test_duplicate_engine.py::test_perceptual_variant_is_review_only
|
||||||
|
delta: >
|
||||||
|
The donor propagated variant links implicitly while writing rows; the replacement makes the canonical link a reviewable, reversible cluster decision, so a propagated link can always be undone.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: pa-hashing
|
- id: pa-hashing
|
||||||
area: hashing
|
area: hashing
|
||||||
@@ -141,8 +153,12 @@ rows:
|
|||||||
classification: replace
|
classification: replace
|
||||||
rationale: Console report; superseded by the duplicate-review API/UI (US01-06).
|
rationale: Console report; superseded by the duplicate-review API/UI (US01-06).
|
||||||
target: photo_pipeline/api/routes + frontend duplicate review
|
target: photo_pipeline/api/routes + frontend duplicate review
|
||||||
pending_story: US01-06
|
parity:
|
||||||
status: pending
|
- tests/integration/test_review_api.py::test_clusters_list_and_detail
|
||||||
|
- tests/e2e/test_review_ui.py::test_fuzzy_decision_requires_confirmation
|
||||||
|
delta: >
|
||||||
|
The text listing became the paged cluster API and the comparison UI; no textual report is produced.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: pa-reconcile
|
- id: pa-reconcile
|
||||||
area: database
|
area: database
|
||||||
@@ -202,7 +218,7 @@ rows:
|
|||||||
RGB-normalize (drops alpha, converts HEIC), LANCZOS resize to 2048px
|
RGB-normalize (drops alpha, converts HEIC), LANCZOS resize to 2048px
|
||||||
long-edge, JPEG q85 base64 — the provider-input contract. Truncated-image
|
long-edge, JPEG q85 base64 — the provider-input contract. Truncated-image
|
||||||
tolerance (ImageFile.LOAD_TRUNCATED_IMAGES) carries with it.
|
tolerance (ImageFile.LOAD_TRUNCATED_IMAGES) carries with it.
|
||||||
target: photo_pipeline/integrations/vision.py
|
target: photo_pipeline/services/analysis.py
|
||||||
tests:
|
tests:
|
||||||
- test_pa_imaging::test_prepare_image_small_passthrough_jpeg
|
- test_pa_imaging::test_prepare_image_small_passthrough_jpeg
|
||||||
- test_pa_imaging::test_prepare_image_resizes_to_max_long_edge
|
- test_pa_imaging::test_prepare_image_resizes_to_max_long_edge
|
||||||
@@ -230,9 +246,12 @@ rows:
|
|||||||
response validation, 429/503 retry with exponential backoff. Prompt and
|
response validation, 429/503 retry with exponential backoff. Prompt and
|
||||||
model/config version must be persisted per analysis_runs. Characterized
|
model/config version must be persisted per analysis_runs. Characterized
|
||||||
against a deterministic fake provider when the analysis service is ported.
|
against a deterministic fake provider when the analysis service is ported.
|
||||||
target: photo_pipeline/integrations/vision.py
|
target: photo_pipeline/services/analysis.py
|
||||||
pending_story: US02-06
|
parity:
|
||||||
status: pending
|
- tests/integration/test_safety_analysis.py::test_provider_called_only_for_confirmed_sfw
|
||||||
|
delta: >
|
||||||
|
The prompt and response schema carry over; the provider is an injected adapter so the privacy gate is testable, and results are keyed to asset ids rather than paths.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: pa-throttle
|
- id: pa-throttle
|
||||||
area: logging
|
area: logging
|
||||||
@@ -241,8 +260,8 @@ rows:
|
|||||||
rationale: >
|
rationale: >
|
||||||
Rolling throttle window + persistent throttle_events.jsonl + RPD day
|
Rolling throttle window + persistent throttle_events.jsonl + RPD day
|
||||||
counter become job metrics/events on the durable job model.
|
counter become job metrics/events on the durable job model.
|
||||||
target: photo_pipeline/jobs/coordinator.py
|
target: photo_pipeline/services/jobs.py + photo_pipeline/services/analysis.py
|
||||||
pending_story: US02-02
|
pending_story: US07-04
|
||||||
status: pending
|
status: pending
|
||||||
|
|
||||||
- id: pa-nsfw-filter
|
- id: pa-nsfw-filter
|
||||||
@@ -312,8 +331,12 @@ rows:
|
|||||||
checked between items, double-SIGINT force quit — becomes the durable
|
checked between items, double-SIGINT force quit — becomes the durable
|
||||||
JobRunner worker loop with the same drain-and-resume semantics.
|
JobRunner worker loop with the same drain-and-resume semantics.
|
||||||
target: photo_pipeline/jobs/worker.py
|
target: photo_pipeline/jobs/worker.py
|
||||||
pending_story: US02-02
|
parity:
|
||||||
status: pending
|
- tests/integration/test_worker.py::test_worker_processes_all_items
|
||||||
|
- tests/integration/test_worker.py::test_cooperative_cancellation_leaves_items_resumable
|
||||||
|
delta: >
|
||||||
|
The in-process folder loop with SIGINT handling became durable jobs claimed by a worker: cancellation is a persisted request, not a signal, and an interrupted run resumes from the database instead of restarting.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: pa-ui-terminal
|
- id: pa-ui-terminal
|
||||||
area: ui
|
area: ui
|
||||||
@@ -371,8 +394,12 @@ rows:
|
|||||||
the JSONL history logger become structured JSON logging with job_id/
|
the JSONL history logger become structured JSON logging with job_id/
|
||||||
asset_id and job_events rows; per-photo history maps to job events.
|
asset_id and job_events rows; per-photo history maps to job events.
|
||||||
target: photo_pipeline structured logging + jobs/job_events
|
target: photo_pipeline structured logging + jobs/job_events
|
||||||
pending_story: US02-02
|
parity:
|
||||||
status: pending
|
- tests/integration/test_jobs.py::test_enqueue_persists_items_and_event
|
||||||
|
- tests/integration/test_jobs_sse.py::test_sse_streams_all_events_then_closes
|
||||||
|
delta: >
|
||||||
|
The JSONL history file and rich console handler are replaced by structured JSON logs plus durable job_events; the browser reads events over SSE rather than tailing a file.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: pa-balance
|
- id: pa-balance
|
||||||
area: vision
|
area: vision
|
||||||
@@ -382,8 +409,11 @@ rows:
|
|||||||
Provider balance/quota probes (report 'unsupported' on providers without
|
Provider balance/quota probes (report 'unsupported' on providers without
|
||||||
the endpoint). Network-bound; characterized against the fake provider.
|
the endpoint). Network-bound; characterized against the fake provider.
|
||||||
target: photo_pipeline/services/analysis.py
|
target: photo_pipeline/services/analysis.py
|
||||||
pending_story: US02-06
|
parity:
|
||||||
status: pending
|
- tests/integration/test_safety_analysis.py::test_provider_called_only_for_confirmed_sfw
|
||||||
|
delta: >
|
||||||
|
Not carried over: balance/quota polling was provider-specific (Gemini/OpenAI billing endpoints) and key-scoped. Cost reporting, when a story asks for it, comes from the per-run usage recorded with each analysis result rather than from a vendor endpoint.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: pa-cli
|
- id: pa-cli
|
||||||
area: configuration
|
area: configuration
|
||||||
@@ -393,9 +423,13 @@ rows:
|
|||||||
argparse surface is superseded by the API; flags map to job configs
|
argparse surface is superseded by the API; flags map to job configs
|
||||||
(documented in WEBAPP_CONCEPT.md §8 parity table). Transitional CLI calls
|
(documented in WEBAPP_CONCEPT.md §8 parity table). Transitional CLI calls
|
||||||
the shared services until archival (E07).
|
the shared services until archival (E07).
|
||||||
target: photo_pipeline/api + transitional CLI
|
target: photo_pipeline/__main__.py (serve | worker | migrate | import-legacy-scores) + /api/v1
|
||||||
pending_story: US07-01
|
parity:
|
||||||
status: pending
|
- tests/integration/test_app_lifecycle.py::test_restart_preserves_data_and_reruns_migrations
|
||||||
|
- tests/unit/test_legacy_archive.py::test_production_code_never_imports_an_archived_module
|
||||||
|
delta: >
|
||||||
|
The argparse surface is not reproduced. Every flag that drove work became an API command or a job configuration; the CLI keeps only what an application needs to be operated (serve, worker, migrate) plus the one-off legacy CSV import.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
# ── nsfwtag/ ───────────────────────────────────────────────────────────────
|
# ── nsfwtag/ ───────────────────────────────────────────────────────────────
|
||||||
- id: nt-discovery
|
- id: nt-discovery
|
||||||
@@ -422,9 +456,17 @@ rows:
|
|||||||
nsfw_scores.csv stops being the source of truth (concept: DB state).
|
nsfw_scores.csv stops being the source of truth (concept: DB state).
|
||||||
Format characterized (4-decimal scores, bad rows dropped) because the
|
Format characterized (4-decimal scores, bad rows dropped) because the
|
||||||
existing CSV must migrate into assets.safety_score.
|
existing CSV must migrate into assets.safety_score.
|
||||||
target: photo_pipeline/repositories (safety), CSV import in US01-02 migration
|
target: photo_pipeline/services/legacy_import.py (one-off import into safety_reviews)
|
||||||
tests: [test_nsfwtag::test_score_cache_roundtrip_and_tolerance]
|
tests: [test_nsfwtag::test_score_cache_roundtrip_and_tolerance]
|
||||||
status: characterized
|
parity:
|
||||||
|
- tests/integration/test_legacy_import.py::test_scores_are_imported_onto_asset_identity
|
||||||
|
- tests/integration/test_legacy_import.py::test_a_reviewed_asset_is_never_overwritten_by_the_csv
|
||||||
|
delta: >
|
||||||
|
The CSV is no longer read at runtime at all: it is imported once into
|
||||||
|
scored-but-unreviewed safety_reviews rows and left on disk untouched. A path
|
||||||
|
that matches nothing is reported, never turned into an asset, and a human
|
||||||
|
decision always outranks an imported score.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: nt-score-model
|
- id: nt-score-model
|
||||||
area: nsfw
|
area: nsfw
|
||||||
@@ -485,8 +527,11 @@ rows:
|
|||||||
Newline-list bulk tagging (nsfw_confirmed.txt flow) is superseded by DB
|
Newline-list bulk tagging (nsfw_confirmed.txt flow) is superseded by DB
|
||||||
review decisions; the existing list is a one-time migration input.
|
review decisions; the existing list is a one-time migration input.
|
||||||
target: photo_pipeline/services/safety.py (decision import in US01-02)
|
target: photo_pipeline/services/safety.py (decision import in US01-02)
|
||||||
pending_story: US01-02
|
parity:
|
||||||
status: pending
|
- tests/integration/test_safety_parity.py::test_extracted_marks_and_partition_match_donor
|
||||||
|
delta: >
|
||||||
|
Bulk keyword application from a file list is replaced by decisions against asset ids; the EXIF write itself is the extracted, read-back-verified one.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: nt-ui
|
- id: nt-ui
|
||||||
area: ui
|
area: ui
|
||||||
@@ -498,8 +543,11 @@ rows:
|
|||||||
threshold/score review flow, lightbox and keyboard model are the frontend
|
threshold/score review flow, lightbox and keyboard model are the frontend
|
||||||
donor for the Safety view (preserved per concept §10; ported in US02-01).
|
donor for the Safety view (preserved per concept §10; ported in US02-01).
|
||||||
target: photo_pipeline/api + frontend Safety view
|
target: photo_pipeline/api + frontend Safety view
|
||||||
pending_story: US02-01
|
parity:
|
||||||
status: pending
|
- tests/e2e/test_workflow_views.py::test_safety_review_decide_persists_across_reload
|
||||||
|
delta: >
|
||||||
|
The stdlib review server is replaced by the API plus the Safety view; the donor's layout, thresholds, and keyboard flow carry over, its embedded HTML generation does not.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: nt-bench
|
- id: nt-bench
|
||||||
area: nsfw
|
area: nsfw
|
||||||
@@ -509,8 +557,11 @@ rows:
|
|||||||
Dev-only model benchmark; archived without webapp replacement (recorded
|
Dev-only model benchmark; archived without webapp replacement (recorded
|
||||||
basis of the AdamCodd model choice). No production caller.
|
basis of the AdamCodd model choice). No production caller.
|
||||||
target: none (archive as reference)
|
target: none (archive as reference)
|
||||||
pending_story: US07-01
|
parity:
|
||||||
status: pending
|
- tests/unit/test_legacy_archive.py::test_every_archived_source_matches_its_checksum
|
||||||
|
delta: >
|
||||||
|
No replacement: a dev-only benchmark whose result (the AdamCodd model choice) is already recorded. Kept in the archive as the basis of that choice.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
# ── webapp/ ────────────────────────────────────────────────────────────────
|
# ── webapp/ ────────────────────────────────────────────────────────────────
|
||||||
- id: wa-query-search
|
- id: wa-query-search
|
||||||
@@ -579,9 +630,13 @@ rows:
|
|||||||
Subprocess-driving-the-CLI job control is superseded by durable DB jobs
|
Subprocess-driving-the-CLI job control is superseded by durable DB jobs
|
||||||
with a worker process. Two ideas carry over: progress derived from DB
|
with a worker process. Two ideas carry over: progress derived from DB
|
||||||
counts (not job-private state) and single-mutating-job enforcement.
|
counts (not job-private state) and single-mutating-job enforcement.
|
||||||
target: photo_pipeline/jobs/coordinator.py
|
target: photo_pipeline/services/jobs.py + photo_pipeline/jobs/worker.py
|
||||||
pending_story: US02-02
|
parity:
|
||||||
status: pending
|
- tests/integration/test_jobs.py::test_idempotency_key_returns_same_job
|
||||||
|
- tests/integration/test_worker.py::test_handler_failure_fails_the_job
|
||||||
|
delta: >
|
||||||
|
Subprocess supervision of a CLI is replaced by durable jobs in the same process family: there is no subprocess to supervise, and progress is persisted rather than scraped from stdout.
|
||||||
|
status: resolved
|
||||||
|
|
||||||
- id: wa-server
|
- id: wa-server
|
||||||
area: ui
|
area: ui
|
||||||
@@ -593,5 +648,9 @@ rows:
|
|||||||
browser. analyzer.html + page.py design (dark OLED tokens, Library/
|
browser. analyzer.html + page.py design (dark OLED tokens, Library/
|
||||||
Analyze/Stats views) is frontend donor material per concept §10.
|
Analyze/Stats views) is frontend donor material per concept §10.
|
||||||
target: photo_pipeline/api/app.py + frontend
|
target: photo_pipeline/api/app.py + frontend
|
||||||
pending_story: US02-05
|
parity:
|
||||||
status: pending
|
- tests/e2e/test_frontend_shell.py::test_shell_loads_assets_without_console_or_network_errors
|
||||||
|
- tests/e2e/test_frontend_shell.py::test_deep_link_and_reload_restore_view_and_filters
|
||||||
|
delta: >
|
||||||
|
The stdlib HTTP routes become the versioned FastAPI API and the static frontend shell; Python no longer interpolates HTML.
|
||||||
|
status: resolved
|
||||||
18
legacy_cli_archive/photo_analyzer.env.sample
Normal file
18
legacy_cli_archive/photo_analyzer.env.sample
Normal file
@@ -0,0 +1,18 @@
|
|||||||
|
# Sample configuration for the archived photo_analyzer.py CLI (US07-01).
|
||||||
|
#
|
||||||
|
# REDACTED: no value below is real. Each line names a variable the CLI read and
|
||||||
|
# what belongs there; the placeholders are deliberately not key-shaped, so this
|
||||||
|
# file can never be mistaken for — or scanned as — a credential.
|
||||||
|
|
||||||
|
LLM_API_KEY=<paste your provider key here>
|
||||||
|
LLM_BASE_URL=<provider base url, e.g. the OpenAI-compatible Gemini endpoint>
|
||||||
|
LLM_MODEL=<model id, e.g. a Gemini Flash release>
|
||||||
|
|
||||||
|
# Optional tuning the CLI read from the same file:
|
||||||
|
PHASH_THRESHOLD=8
|
||||||
|
MAX_WORKERS=4
|
||||||
|
RETRY_ATTEMPTS=3
|
||||||
|
RPD_LIMIT=0
|
||||||
|
|
||||||
|
# The CLI ignored its own shipped placeholder (a literal "sk-REPLACE..." string)
|
||||||
|
# until it was replaced, and a shell variable always won over this file.
|
||||||
20
legacy_cli_archive/requirements-lock.txt
Normal file
20
legacy_cli_archive/requirements-lock.txt
Normal file
@@ -0,0 +1,20 @@
|
|||||||
|
# Final dependency lock of the archived CLIs (US07-01).
|
||||||
|
#
|
||||||
|
# These are the versions present in the environment the archive was taken from —
|
||||||
|
# what the frozen sources were last verified against by the characterization
|
||||||
|
# suite. Restoring a donor for forensics means pinning these, not "latest".
|
||||||
|
#
|
||||||
|
# Python 3.14.6
|
||||||
|
|
||||||
|
openai==3.0.0 # photo_analyzer: OpenAI-compatible vision client
|
||||||
|
numpy==2.4.6 # photo_analyzer, nsfwtag: pixel work
|
||||||
|
Pillow==12.3.0 # photo_analyzer, nsfwtag: decode/resize
|
||||||
|
rich==15.0.0 # photo_analyzer: console output
|
||||||
|
scipy==1.18.0 # photo_analyzer: perceptual-hash DCT
|
||||||
|
PyYAML==6.0.3 # tooling that reads the donor ledger
|
||||||
|
|
||||||
|
# NSFW inference (nsfwtag/scoring.py, nsfwtag/bench.py) was never installed in the
|
||||||
|
# archiving environment; the model stack is recorded here from the sources so a
|
||||||
|
# forensic run can reproduce it, not from a resolved lock:
|
||||||
|
# torch, transformers, timm — AdamCodd/vit-base-nsfw-detector (see nsfwtag/README.md)
|
||||||
|
# exiftool is an external binary, not a Python package.
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
"""Application management CLI: ``python -m photo_pipeline {serve,migrate}``."""
|
"""Application management CLI: ``python -m photo_pipeline {serve,migrate,worker,import-legacy-scores}``."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -16,6 +16,17 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|||||||
commands.add_parser("migrate", help="Upgrade the database to the latest revision")
|
commands.add_parser("migrate", help="Upgrade the database to the latest revision")
|
||||||
worker_cmd = commands.add_parser("worker", help="Run a durable-job worker")
|
worker_cmd = commands.add_parser("worker", help="Run a durable-job worker")
|
||||||
worker_cmd.add_argument("--id", default="worker-1", help="Worker id (lease owner)")
|
worker_cmd.add_argument("--id", default="worker-1", help="Worker id (lease owner)")
|
||||||
|
import_cmd = commands.add_parser(
|
||||||
|
"import-legacy-scores",
|
||||||
|
help="Import the archived CLI's nsfw_scores.csv into the database (US07-01)",
|
||||||
|
)
|
||||||
|
import_cmd.add_argument("csv", help="Path to nsfw_scores.csv")
|
||||||
|
import_cmd.add_argument(
|
||||||
|
"--overwrite", action="store_true", help="Replace differing imported scores"
|
||||||
|
)
|
||||||
|
import_cmd.add_argument(
|
||||||
|
"--dry-run", action="store_true", help="Report what would happen and change nothing"
|
||||||
|
)
|
||||||
args = parser.parse_args(argv)
|
args = parser.parse_args(argv)
|
||||||
|
|
||||||
config = Config.from_env()
|
config = Config.from_env()
|
||||||
@@ -25,6 +36,24 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|||||||
run_migrations(config.database_url)
|
run_migrations(config.database_url)
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
if args.command == "import-legacy-scores":
|
||||||
|
import json
|
||||||
|
|
||||||
|
from photo_pipeline.db import create_db_engine, create_session_factory
|
||||||
|
from photo_pipeline.services.legacy_import import LegacyImportService, write_report
|
||||||
|
|
||||||
|
run_migrations(config.database_url)
|
||||||
|
engine = create_db_engine(config.database_url)
|
||||||
|
service = LegacyImportService(create_session_factory(engine))
|
||||||
|
report = service.import_nsfw_scores(
|
||||||
|
args.csv, overwrite=args.overwrite, dry_run=args.dry_run
|
||||||
|
)
|
||||||
|
# The report is the point: an import nobody can audit is not a migration.
|
||||||
|
if not args.dry_run:
|
||||||
|
write_report(report, config.data_dir)
|
||||||
|
print(json.dumps(report.counts, indent=2))
|
||||||
|
return 0
|
||||||
|
|
||||||
if args.command == "worker":
|
if args.command == "worker":
|
||||||
from photo_pipeline.db import create_db_engine, create_session_factory
|
from photo_pipeline.db import create_db_engine, create_session_factory
|
||||||
|
|
||||||
|
|||||||
220
photo_pipeline/services/legacy_import.py
Normal file
220
photo_pipeline/services/legacy_import.py
Normal file
@@ -0,0 +1,220 @@
|
|||||||
|
"""Import the last path-keyed CSV state into the database (US07-01).
|
||||||
|
|
||||||
|
``nsfwtag`` cached its safety scores in ``nsfw_scores.csv`` next to the library:
|
||||||
|
one ``path,nsfw_score`` row per photo, four decimals, unreadable rows dropped. That
|
||||||
|
file stops being a source of truth when the CLI is archived, so its scores are
|
||||||
|
imported once — as scored-but-unreviewed ``safety_reviews`` rows on the stable
|
||||||
|
``assets.id`` each path resolves to — and the CSV is left untouched on disk as its
|
||||||
|
own backup.
|
||||||
|
|
||||||
|
The import is deliberately conservative, because a score is evidence about a photo
|
||||||
|
and a path is not an identity:
|
||||||
|
|
||||||
|
- a row whose path matches no known asset is **unmatched**, never a new asset;
|
||||||
|
- an asset a human already reviewed is **never** touched: a score is evidence, a
|
||||||
|
decision is a judgement, and an import may not overwrite the second with the
|
||||||
|
first — the difference is reported as a conflict instead;
|
||||||
|
- a row for an asset that already carries a score is **skipped** unless
|
||||||
|
``overwrite`` is asked for, and a differing score is reported either way;
|
||||||
|
- an unparsable score or a duplicate path is **rejected** with its reason;
|
||||||
|
- rerunning changes nothing that is already imported.
|
||||||
|
|
||||||
|
Everything it did — and everything it refused to do — comes back as a
|
||||||
|
reconciliation report, which the caller can persist next to the database. The
|
||||||
|
donor's own reader is the specification for the format (donor ledger:
|
||||||
|
``nt-score-cache``).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import csv
|
||||||
|
import json
|
||||||
|
import uuid
|
||||||
|
from collections import Counter
|
||||||
|
from dataclasses import asdict, dataclass, field
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from sqlalchemy import select
|
||||||
|
from sqlalchemy.orm import sessionmaker
|
||||||
|
|
||||||
|
from photo_pipeline.models import Asset, AssetPath, SafetyReview
|
||||||
|
|
||||||
|
REPORT_VERSION = 1
|
||||||
|
# The donor wrote scores as four-decimal strings; anything outside 0..1 was never
|
||||||
|
# something it produced, so it is data corruption rather than a score.
|
||||||
|
SCORE_RANGE = (0.0, 1.0)
|
||||||
|
# Recorded as the reviewer so an imported score is never mistaken for a judgement.
|
||||||
|
REVIEWER = "legacy-csv-import"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ImportReport:
|
||||||
|
source: str
|
||||||
|
rows: int = 0
|
||||||
|
imported: int = 0
|
||||||
|
skipped_existing: int = 0
|
||||||
|
unchanged: int = 0
|
||||||
|
reviewed: int = 0
|
||||||
|
unmatched: list[str] = field(default_factory=list)
|
||||||
|
rejected: list[dict] = field(default_factory=list)
|
||||||
|
conflicts: list[dict] = field(default_factory=list)
|
||||||
|
generated_at: str = ""
|
||||||
|
schema_version: int = REPORT_VERSION
|
||||||
|
|
||||||
|
def to_dict(self) -> dict:
|
||||||
|
return asdict(self)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def counts(self) -> dict[str, int]:
|
||||||
|
return {
|
||||||
|
"rows": self.rows,
|
||||||
|
"imported": self.imported,
|
||||||
|
"skipped_existing": self.skipped_existing,
|
||||||
|
"unchanged": self.unchanged,
|
||||||
|
"reviewed": self.reviewed,
|
||||||
|
"unmatched": len(self.unmatched),
|
||||||
|
"rejected": len(self.rejected),
|
||||||
|
"conflicts": len(self.conflicts),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class LegacyImportService:
|
||||||
|
def __init__(self, session_factory: sessionmaker) -> None:
|
||||||
|
self._session_factory = session_factory
|
||||||
|
|
||||||
|
def import_nsfw_scores(
|
||||||
|
self, csv_path: Path | str, *, overwrite: bool = False, dry_run: bool = False
|
||||||
|
) -> ImportReport:
|
||||||
|
"""Import ``nsfw_scores.csv`` onto asset identity and report what happened."""
|
||||||
|
path = Path(csv_path)
|
||||||
|
report = ImportReport(source=str(path))
|
||||||
|
if not path.is_file():
|
||||||
|
report.rejected.append({"path": str(path), "reason": "csv_missing"})
|
||||||
|
return self._stamp(report)
|
||||||
|
|
||||||
|
with self._session_factory() as session:
|
||||||
|
# Paths are matched against every occurrence an asset ever had, so a
|
||||||
|
# photo scored before a rename is still recognised.
|
||||||
|
by_path = self._path_index(session)
|
||||||
|
latest = self._latest_reviews(session)
|
||||||
|
seen: Counter[str] = Counter()
|
||||||
|
|
||||||
|
for row in self._rows(path, report):
|
||||||
|
report.rows += 1
|
||||||
|
raw_path, raw_score = row
|
||||||
|
seen[raw_path] += 1
|
||||||
|
if seen[raw_path] > 1:
|
||||||
|
report.rejected.append({"path": raw_path, "reason": "duplicate_path"})
|
||||||
|
continue
|
||||||
|
score = _parse_score(raw_score)
|
||||||
|
if score is None:
|
||||||
|
report.rejected.append(
|
||||||
|
{"path": raw_path, "reason": "unparsable_score", "value": raw_score}
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
asset_id = by_path.get(raw_path)
|
||||||
|
if asset_id is None:
|
||||||
|
report.unmatched.append(raw_path)
|
||||||
|
continue
|
||||||
|
|
||||||
|
current = latest.get(asset_id)
|
||||||
|
if current is not None and current.decision:
|
||||||
|
# A human decided this one; the CSV is older evidence.
|
||||||
|
report.reviewed += 1
|
||||||
|
if current.score is None or abs(current.score - score) >= 1e-9:
|
||||||
|
report.conflicts.append(
|
||||||
|
{
|
||||||
|
"path": raw_path,
|
||||||
|
"asset_id": asset_id,
|
||||||
|
"current": current.score,
|
||||||
|
"decision": current.decision,
|
||||||
|
"csv": score,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
if current is not None and current.score is not None:
|
||||||
|
if abs(current.score - score) < 1e-9:
|
||||||
|
report.unchanged += 1
|
||||||
|
continue
|
||||||
|
report.conflicts.append(
|
||||||
|
{
|
||||||
|
"path": raw_path,
|
||||||
|
"asset_id": asset_id,
|
||||||
|
"current": current.score,
|
||||||
|
"decision": None,
|
||||||
|
"csv": score,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
if not overwrite:
|
||||||
|
report.skipped_existing += 1
|
||||||
|
continue
|
||||||
|
session.add(
|
||||||
|
SafetyReview(
|
||||||
|
id=str(uuid.uuid4()),
|
||||||
|
asset_id=asset_id,
|
||||||
|
score=score,
|
||||||
|
reviewer=REVIEWER,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
latest[asset_id] = SafetyReview(asset_id=asset_id, score=score)
|
||||||
|
report.imported += 1
|
||||||
|
|
||||||
|
if dry_run:
|
||||||
|
session.rollback()
|
||||||
|
else:
|
||||||
|
session.commit()
|
||||||
|
return self._stamp(report)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _rows(path: Path, report: ImportReport):
|
||||||
|
"""Yield ``(path, score)`` pairs, tolerating the donor's own sloppiness."""
|
||||||
|
with open(path, newline="", encoding="utf-8", errors="replace") as handle:
|
||||||
|
for row in csv.DictReader(handle):
|
||||||
|
raw_path = (row.get("path") or "").strip()
|
||||||
|
if not raw_path:
|
||||||
|
report.rejected.append({"path": "", "reason": "missing_path"})
|
||||||
|
continue
|
||||||
|
yield raw_path, row.get("nsfw_score")
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _latest_reviews(session) -> dict[str, SafetyReview]:
|
||||||
|
"""The current safety row per asset — latest wins, as everywhere else."""
|
||||||
|
latest: dict[str, SafetyReview] = {}
|
||||||
|
for review in session.scalars(select(SafetyReview).order_by(SafetyReview.created_at)):
|
||||||
|
latest[review.asset_id] = review
|
||||||
|
return latest
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _path_index(session) -> dict[str, str]:
|
||||||
|
"""Every path an asset is or was known by → its stable id."""
|
||||||
|
index: dict[str, str] = {}
|
||||||
|
for asset_id, path in session.execute(select(AssetPath.asset_id, AssetPath.path)):
|
||||||
|
index.setdefault(path, asset_id)
|
||||||
|
for asset_id, path in session.execute(select(Asset.id, Asset.current_path)):
|
||||||
|
if path:
|
||||||
|
index[path] = asset_id # the current path wins over a closed one
|
||||||
|
return index
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _stamp(report: ImportReport) -> ImportReport:
|
||||||
|
report.generated_at = datetime.now(timezone.utc).isoformat()
|
||||||
|
return report
|
||||||
|
|
||||||
|
|
||||||
|
def write_report(report: ImportReport, directory: Path) -> Path:
|
||||||
|
"""Persist the reconciliation report; the import is not evidence until it is."""
|
||||||
|
directory.mkdir(parents=True, exist_ok=True)
|
||||||
|
path = directory / "legacy-nsfw-import.json"
|
||||||
|
path.write_text(json.dumps(report.to_dict(), indent=2, sort_keys=True), encoding="utf-8")
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_score(value) -> float | None:
|
||||||
|
try:
|
||||||
|
score = float(value)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
if not SCORE_RANGE[0] <= score <= SCORE_RANGE[1]:
|
||||||
|
return None
|
||||||
|
return score
|
||||||
@@ -24,6 +24,8 @@ test = [
|
|||||||
|
|
||||||
[tool.ruff]
|
[tool.ruff]
|
||||||
line-length = 100
|
line-length = 100
|
||||||
|
# The frozen CLI archive is evidence, not code under maintenance (US07-01).
|
||||||
|
extend-exclude = ["legacy_cli_archive"]
|
||||||
|
|
||||||
[tool.pytest.ini_options]
|
[tool.pytest.ini_options]
|
||||||
testpaths = ["tests"]
|
testpaths = ["tests"]
|
||||||
|
|||||||
@@ -14,6 +14,11 @@ import pytest
|
|||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
REPO = Path(__file__).resolve().parents[2]
|
REPO = Path(__file__).resolve().parents[2]
|
||||||
|
# The donors are frozen in the read-only archive (US07-01). Only this suite — and
|
||||||
|
# the parity test that compares against them — puts that directory on sys.path;
|
||||||
|
# production never does, which tests/unit/test_legacy_archive.py enforces.
|
||||||
|
ARCHIVED_SOURCES = REPO / "legacy_cli_archive" / "src"
|
||||||
|
sys.path.insert(0, str(ARCHIVED_SOURCES))
|
||||||
sys.path.insert(0, str(REPO))
|
sys.path.insert(0, str(REPO))
|
||||||
|
|
||||||
EXIFTOOL = shutil.which("exiftool")
|
EXIFTOOL = shutil.which("exiftool")
|
||||||
|
|||||||
@@ -1,13 +1,20 @@
|
|||||||
"""Ledger lint (US01-01): every donor-ledger row must carry a real source
|
"""Ledger lint (US01-01, extended by US07-01).
|
||||||
reference, a target location, and either existing characterization test IDs or
|
|
||||||
a real pending backlog story."""
|
Every donor-ledger row must carry a real source reference — now inside the frozen
|
||||||
|
archive — a target location, and either existing characterization test IDs or a
|
||||||
|
real pending backlog story. Since archival (US07-01) a row may also be ``resolved``:
|
||||||
|
its replacement has shipped, ``parity`` names tests that exist and prove it, and
|
||||||
|
``delta`` states every intentional difference. Nothing may quietly become
|
||||||
|
"finished" without one of those two."""
|
||||||
import re
|
import re
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import yaml
|
import yaml
|
||||||
|
|
||||||
REPO = Path(__file__).resolve().parents[2]
|
REPO = Path(__file__).resolve().parents[2]
|
||||||
LEDGER = REPO / "donor_ledger.yaml"
|
ARCHIVE = REPO / "legacy_cli_archive"
|
||||||
|
LEDGER = ARCHIVE / "donor_ledger.yaml"
|
||||||
|
ARCHIVED_SOURCES = ARCHIVE / "src"
|
||||||
STORIES = REPO / "delivery_backlog" / "stories"
|
STORIES = REPO / "delivery_backlog" / "stories"
|
||||||
TESTS_DIR = Path(__file__).resolve().parent
|
TESTS_DIR = Path(__file__).resolve().parent
|
||||||
|
|
||||||
@@ -15,7 +22,7 @@ CLASSIFICATIONS = {"reuse", "extract", "refactor", "replace"}
|
|||||||
REQUIRED_AREAS = {"discovery", "hashing", "imaging", "nsfw", "vision", "exif",
|
REQUIRED_AREAS = {"discovery", "hashing", "imaging", "nsfw", "vision", "exif",
|
||||||
"database", "ui", "configuration", "logging", "cancellation",
|
"database", "ui", "configuration", "logging", "cancellation",
|
||||||
"error"}
|
"error"}
|
||||||
STATUSES = {"characterized", "pending"}
|
STATUSES = {"characterized", "resolved", "pending"}
|
||||||
|
|
||||||
|
|
||||||
def load_rows():
|
def load_rows():
|
||||||
@@ -49,10 +56,11 @@ def test_rows_have_required_fields_and_unique_ids():
|
|||||||
|
|
||||||
|
|
||||||
def test_source_references_resolve():
|
def test_source_references_resolve():
|
||||||
|
"""Source paths are relative to the archive: the donors moved there, whole."""
|
||||||
for r in load_rows():
|
for r in load_rows():
|
||||||
src = r["source"]
|
src = r["source"]
|
||||||
f = REPO / src["file"]
|
f = ARCHIVED_SOURCES / src["file"]
|
||||||
assert f.is_file(), f"{r['id']}: source file {src['file']} missing"
|
assert f.is_file(), f"{r['id']}: source file {src['file']} missing from the archive"
|
||||||
text = f.read_text(encoding="utf-8")
|
text = f.read_text(encoding="utf-8")
|
||||||
for sym in src["symbols"]:
|
for sym in src["symbols"]:
|
||||||
assert sym in text, f"{r['id']}: symbol {sym!r} not found in {src['file']}"
|
assert sym in text, f"{r['id']}: symbol {sym!r} not found in {src['file']}"
|
||||||
@@ -63,7 +71,9 @@ def test_rows_have_tests_or_pending_story():
|
|||||||
for r in load_rows():
|
for r in load_rows():
|
||||||
tests = r.get("tests", [])
|
tests = r.get("tests", [])
|
||||||
pending = r.get("pending_story")
|
pending = r.get("pending_story")
|
||||||
assert tests or pending, f"{r['id']}: neither tests nor pending_story"
|
parity = r.get("parity", [])
|
||||||
|
assert tests or pending or parity, \
|
||||||
|
f"{r['id']}: neither characterization tests, parity tests, nor a pending story"
|
||||||
for t in tests:
|
for t in tests:
|
||||||
assert t in known_tests, f"{r['id']}: unknown test id {t}"
|
assert t in known_tests, f"{r['id']}: unknown test id {t}"
|
||||||
if pending:
|
if pending:
|
||||||
@@ -73,16 +83,58 @@ def test_rows_have_tests_or_pending_story():
|
|||||||
assert tests, f"{r['id']}: characterized rows need test ids"
|
assert tests, f"{r['id']}: characterized rows need test ids"
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolved_rows_name_their_parity_or_their_delta():
|
||||||
|
"""A resolved row is a claim that the behavior is handled. It has to say how:
|
||||||
|
tests that prove the replacement, or a stated difference — usually both."""
|
||||||
|
for r in load_rows():
|
||||||
|
if r["status"] != "resolved":
|
||||||
|
assert "parity" not in r, f"{r['id']}: parity on a non-resolved row"
|
||||||
|
continue
|
||||||
|
parity = r.get("parity", [])
|
||||||
|
delta = r.get("delta")
|
||||||
|
assert parity or delta, f"{r['id']}: resolved without parity tests or a delta"
|
||||||
|
if delta:
|
||||||
|
assert len(str(delta).strip()) >= 20, f"{r['id']}: delta too thin to be a reason"
|
||||||
|
for ref in parity:
|
||||||
|
rel, _, func = ref.partition("::")
|
||||||
|
path = REPO / rel
|
||||||
|
assert path.is_file(), f"{r['id']}: parity test file {rel} missing"
|
||||||
|
assert f"def {func}" in path.read_text(encoding="utf-8"), \
|
||||||
|
f"{r['id']}: parity test {ref} not found"
|
||||||
|
|
||||||
|
|
||||||
|
def test_pending_rows_are_the_only_unfinished_work():
|
||||||
|
"""The ledger is the honest list of what has not been carried over: a pending
|
||||||
|
row names the story that will, and that story must still be open work."""
|
||||||
|
for r in load_rows():
|
||||||
|
if r["status"] != "pending":
|
||||||
|
continue
|
||||||
|
story = r.get("pending_story")
|
||||||
|
assert story, f"{r['id']}: pending without a story"
|
||||||
|
assert list(STORIES.glob(f"{story}-*.md")), f"{r['id']}: unknown story {story}"
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_target_module_exists():
|
||||||
|
"""A row is only finished if the thing it points at is really there."""
|
||||||
|
for r in load_rows():
|
||||||
|
if r["status"] == "pending":
|
||||||
|
continue
|
||||||
|
modules = re.findall(r"photo_pipeline/[\w/]+\.py", str(r["target"]))
|
||||||
|
for module in modules:
|
||||||
|
assert (REPO / module).is_file(), f"{r['id']}: target {module} does not exist"
|
||||||
|
|
||||||
|
|
||||||
def test_all_required_areas_covered():
|
def test_all_required_areas_covered():
|
||||||
covered = {r["area"] for r in load_rows()}
|
covered = {r["area"] for r in load_rows()}
|
||||||
assert REQUIRED_AREAS <= covered, f"uncovered areas: {REQUIRED_AREAS - covered}"
|
assert REQUIRED_AREAS <= covered, f"uncovered areas: {REQUIRED_AREAS - covered}"
|
||||||
assert covered <= REQUIRED_AREAS, f"unknown areas: {covered - REQUIRED_AREAS}"
|
assert covered <= REQUIRED_AREAS, f"unknown areas: {covered - REQUIRED_AREAS}"
|
||||||
|
|
||||||
|
|
||||||
def test_no_legacy_file_moved():
|
def test_every_donor_is_in_the_archive_and_nowhere_else():
|
||||||
# US01-01 explicitly forbids moving/archiving donors; the ledger's source
|
"""US01-01 forbade archiving before characterization; US07-01 requires it after.
|
||||||
# files must all still exist at their original locations.
|
Each donor exists exactly once — frozen, in the archive."""
|
||||||
for donor in ("photo_analyzer.py", "nsfwtag/scoring.py", "nsfwtag/exif.py",
|
for donor in ("photo_analyzer.py", "nsfwtag/scoring.py", "nsfwtag/exif.py",
|
||||||
"nsfwtag/server.py", "webapp/query.py", "webapp/runner.py",
|
"nsfwtag/server.py", "webapp/query.py", "webapp/runner.py",
|
||||||
"webapp/server.py"):
|
"webapp/server.py"):
|
||||||
assert (REPO / donor).is_file(), f"donor moved: {donor}"
|
assert (ARCHIVED_SOURCES / donor).is_file(), f"donor missing from archive: {donor}"
|
||||||
|
assert not (REPO / donor).exists(), f"donor still live at the repo root: {donor}"
|
||||||
|
|||||||
238
tests/integration/test_legacy_import.py
Normal file
238
tests/integration/test_legacy_import.py
Normal file
@@ -0,0 +1,238 @@
|
|||||||
|
"""Importing the archived CLI's CSV state into the database (US07-01).
|
||||||
|
|
||||||
|
The donor's ``nsfw_scores.csv`` was keyed by path; the database is keyed by a
|
||||||
|
stable asset id. Every case here is about that gap: a path that moved, a path that
|
||||||
|
matches nothing, a score a human has already overruled, and a file that is simply
|
||||||
|
malformed. The import may add evidence and must never invent an asset, overwrite a
|
||||||
|
judgement, or fail silently — whatever it does ends up in the report.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import uuid
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from sqlalchemy import select
|
||||||
|
|
||||||
|
from photo_pipeline.db import create_db_engine, create_session_factory, run_migrations
|
||||||
|
from photo_pipeline.models import Asset, AssetPath, SafetyReview
|
||||||
|
from photo_pipeline.services.legacy_import import (
|
||||||
|
REVIEWER,
|
||||||
|
LegacyImportService,
|
||||||
|
write_report,
|
||||||
|
)
|
||||||
|
|
||||||
|
NOW = datetime(2026, 1, 1, tzinfo=timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def factory(tmp_path):
|
||||||
|
url = f"sqlite:///{tmp_path / 'legacy.db'}"
|
||||||
|
run_migrations(url)
|
||||||
|
engine = create_db_engine(url)
|
||||||
|
yield create_session_factory(engine)
|
||||||
|
engine.dispose()
|
||||||
|
|
||||||
|
|
||||||
|
def _asset(sf, path: Path, *, previous: Path | None = None) -> str:
|
||||||
|
asset_id = str(uuid.uuid4())
|
||||||
|
with sf() as session:
|
||||||
|
session.add(
|
||||||
|
Asset(
|
||||||
|
id=asset_id,
|
||||||
|
original_path=str(previous or path),
|
||||||
|
current_path=str(path),
|
||||||
|
discovered_at=NOW,
|
||||||
|
hash_version=1,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
session.add(AssetPath(asset_id=asset_id, path=str(path), valid_from=NOW))
|
||||||
|
if previous is not None:
|
||||||
|
session.add(
|
||||||
|
AssetPath(
|
||||||
|
asset_id=asset_id, path=str(previous), valid_from=NOW, valid_until=NOW
|
||||||
|
)
|
||||||
|
)
|
||||||
|
session.commit()
|
||||||
|
return asset_id
|
||||||
|
|
||||||
|
|
||||||
|
def _review(sf, asset_id: str, *, score=None, decision=None, at=NOW) -> None:
|
||||||
|
with sf() as session:
|
||||||
|
session.add(
|
||||||
|
SafetyReview(
|
||||||
|
id=str(uuid.uuid4()),
|
||||||
|
asset_id=asset_id,
|
||||||
|
score=score,
|
||||||
|
decision=decision,
|
||||||
|
reviewer="dom",
|
||||||
|
created_at=at,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
session.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def _csv(tmp_path, rows: str) -> Path:
|
||||||
|
path = tmp_path / "nsfw_scores.csv"
|
||||||
|
path.write_text("path,nsfw_score\n" + rows, encoding="utf-8")
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def _scores(sf) -> dict[str, float]:
|
||||||
|
with sf() as session:
|
||||||
|
return {
|
||||||
|
review.asset_id: review.score
|
||||||
|
for review in session.scalars(
|
||||||
|
select(SafetyReview).order_by(SafetyReview.created_at)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_scores_are_imported_onto_asset_identity(tmp_path, factory):
|
||||||
|
a = _asset(factory, tmp_path / "a.jpg")
|
||||||
|
# This one was scored under its old path and has since moved.
|
||||||
|
b = _asset(factory, tmp_path / "moved" / "b.jpg", previous=tmp_path / "b.jpg")
|
||||||
|
csv_path = _csv(
|
||||||
|
tmp_path, f"{tmp_path / 'a.jpg'},0.9123\n{tmp_path / 'b.jpg'},0.0100\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
report = LegacyImportService(factory).import_nsfw_scores(csv_path)
|
||||||
|
|
||||||
|
assert report.counts == {
|
||||||
|
"rows": 2,
|
||||||
|
"imported": 2,
|
||||||
|
"skipped_existing": 0,
|
||||||
|
"unchanged": 0,
|
||||||
|
"reviewed": 0,
|
||||||
|
"unmatched": 0,
|
||||||
|
"rejected": 0,
|
||||||
|
"conflicts": 0,
|
||||||
|
}
|
||||||
|
assert _scores(factory) == {a: 0.9123, b: 0.0100}
|
||||||
|
# The imported rows are evidence, not judgements: no decision is invented.
|
||||||
|
with factory() as session:
|
||||||
|
rows = list(session.scalars(select(SafetyReview)))
|
||||||
|
assert {r.decision for r in rows} == {None}
|
||||||
|
assert {r.reviewer for r in rows} == {REVIEWER}
|
||||||
|
assert csv_path.exists(), "the CSV is left on disk as its own backup"
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_reviewed_asset_is_never_overwritten_by_the_csv(tmp_path, factory):
|
||||||
|
asset_id = _asset(factory, tmp_path / "a.jpg")
|
||||||
|
_review(factory, asset_id, score=0.2, decision="sfw")
|
||||||
|
csv_path = _csv(tmp_path, f"{tmp_path / 'a.jpg'},0.9999\n")
|
||||||
|
|
||||||
|
report = LegacyImportService(factory).import_nsfw_scores(csv_path, overwrite=True)
|
||||||
|
|
||||||
|
assert (report.imported, report.reviewed) == (0, 1)
|
||||||
|
assert report.conflicts[0]["decision"] == "sfw"
|
||||||
|
assert report.conflicts[0]["csv"] == 0.9999
|
||||||
|
with factory() as session:
|
||||||
|
rows = list(session.scalars(select(SafetyReview)))
|
||||||
|
assert len(rows) == 1 and rows[0].decision == "sfw"
|
||||||
|
|
||||||
|
|
||||||
|
def test_an_unknown_path_is_reported_never_turned_into_an_asset(tmp_path, factory):
|
||||||
|
_asset(factory, tmp_path / "a.jpg")
|
||||||
|
csv_path = _csv(
|
||||||
|
tmp_path, f"{tmp_path / 'a.jpg'},0.5000\n{tmp_path / 'ghost.jpg'},0.5000\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
report = LegacyImportService(factory).import_nsfw_scores(csv_path)
|
||||||
|
|
||||||
|
assert report.unmatched == [str(tmp_path / "ghost.jpg")]
|
||||||
|
with factory() as session:
|
||||||
|
assert session.scalar(select(Asset).where(Asset.current_path.like("%ghost%"))) is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_malformed_rows_are_rejected_with_their_reason(tmp_path, factory):
|
||||||
|
_asset(factory, tmp_path / "a.jpg")
|
||||||
|
csv_path = _csv(
|
||||||
|
tmp_path,
|
||||||
|
f"{tmp_path / 'a.jpg'},0.5000\n"
|
||||||
|
f"{tmp_path / 'a.jpg'},0.5000\n" # the same path twice
|
||||||
|
f"{tmp_path / 'b.jpg'},not-a-number\n"
|
||||||
|
f"{tmp_path / 'c.jpg'},7.5\n" # outside 0..1: corruption, not a score
|
||||||
|
",0.5\n", # no path at all
|
||||||
|
)
|
||||||
|
|
||||||
|
report = LegacyImportService(factory).import_nsfw_scores(csv_path)
|
||||||
|
|
||||||
|
reasons = sorted(r["reason"] for r in report.rejected)
|
||||||
|
assert reasons == ["duplicate_path", "missing_path", "unparsable_score", "unparsable_score"]
|
||||||
|
assert report.imported == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_rerunning_changes_nothing_and_a_differing_score_needs_overwrite(tmp_path, factory):
|
||||||
|
asset_id = _asset(factory, tmp_path / "a.jpg")
|
||||||
|
csv_path = _csv(tmp_path, f"{tmp_path / 'a.jpg'},0.5000\n")
|
||||||
|
service = LegacyImportService(factory)
|
||||||
|
service.import_nsfw_scores(csv_path)
|
||||||
|
|
||||||
|
again = service.import_nsfw_scores(csv_path)
|
||||||
|
assert (again.imported, again.unchanged) == (0, 1)
|
||||||
|
assert len(_scores(factory)) == 1
|
||||||
|
|
||||||
|
changed = _csv(tmp_path, f"{tmp_path / 'a.jpg'},0.8000\n")
|
||||||
|
refused = service.import_nsfw_scores(changed)
|
||||||
|
assert (refused.imported, refused.skipped_existing) == (0, 1)
|
||||||
|
assert refused.conflicts[0]["current"] == 0.5
|
||||||
|
|
||||||
|
forced = service.import_nsfw_scores(changed, overwrite=True)
|
||||||
|
assert forced.imported == 1
|
||||||
|
with factory() as session:
|
||||||
|
latest = list(session.scalars(select(SafetyReview).order_by(SafetyReview.created_at)))[-1]
|
||||||
|
assert (latest.asset_id, latest.score) == (asset_id, 0.8)
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_dry_run_reports_without_writing(tmp_path, factory):
|
||||||
|
_asset(factory, tmp_path / "a.jpg")
|
||||||
|
csv_path = _csv(tmp_path, f"{tmp_path / 'a.jpg'},0.5000\n")
|
||||||
|
|
||||||
|
report = LegacyImportService(factory).import_nsfw_scores(csv_path, dry_run=True)
|
||||||
|
|
||||||
|
assert report.imported == 1
|
||||||
|
assert _scores(factory) == {}, "a dry run must leave the database alone"
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_missing_csv_is_a_reported_outcome_not_a_crash(tmp_path, factory):
|
||||||
|
report = LegacyImportService(factory).import_nsfw_scores(tmp_path / "nothing.csv")
|
||||||
|
assert report.rejected == [{"path": str(tmp_path / "nothing.csv"), "reason": "csv_missing"}]
|
||||||
|
assert report.rows == 0
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_report_is_written_where_it_can_be_audited(tmp_path, factory):
|
||||||
|
_asset(factory, tmp_path / "a.jpg")
|
||||||
|
csv_path = _csv(tmp_path, f"{tmp_path / 'a.jpg'},0.5000\n")
|
||||||
|
report = LegacyImportService(factory).import_nsfw_scores(csv_path)
|
||||||
|
|
||||||
|
written = write_report(report, tmp_path / "data")
|
||||||
|
|
||||||
|
payload = json.loads(written.read_text(encoding="utf-8"))
|
||||||
|
assert payload["imported"] == 1
|
||||||
|
assert payload["source"] == str(csv_path)
|
||||||
|
assert payload["schema_version"] == 1
|
||||||
|
assert payload["generated_at"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_later_review_still_wins_after_an_import(tmp_path, factory):
|
||||||
|
"""The import is a floor, not a ceiling: a human decision made afterwards is
|
||||||
|
the current state, exactly as everywhere else in the app."""
|
||||||
|
asset_id = _asset(factory, tmp_path / "a.jpg")
|
||||||
|
csv_path = _csv(tmp_path, f"{tmp_path / 'a.jpg'},0.9000\n")
|
||||||
|
LegacyImportService(factory).import_nsfw_scores(csv_path)
|
||||||
|
# The import stamps itself with the wall clock, so "afterwards" has to be too.
|
||||||
|
_review(
|
||||||
|
factory,
|
||||||
|
asset_id,
|
||||||
|
score=0.9,
|
||||||
|
decision="sfw",
|
||||||
|
at=datetime.now(timezone.utc) + timedelta(hours=1),
|
||||||
|
)
|
||||||
|
|
||||||
|
with factory() as session:
|
||||||
|
latest = list(session.scalars(select(SafetyReview).order_by(SafetyReview.created_at)))[-1]
|
||||||
|
assert latest.decision == "sfw"
|
||||||
@@ -3,6 +3,8 @@ same fixture files (real exiftool-written keywords)."""
|
|||||||
|
|
||||||
import shutil
|
import shutil
|
||||||
import subprocess
|
import subprocess
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import pytest
|
import pytest
|
||||||
@@ -11,6 +13,10 @@ from PIL import Image
|
|||||||
from photo_pipeline.integrations import exiftool
|
from photo_pipeline.integrations import exiftool
|
||||||
from photo_pipeline.services import safety
|
from photo_pipeline.services import safety
|
||||||
|
|
||||||
|
# The donor is frozen in the read-only archive (US07-01); parity still compares
|
||||||
|
# against it, so this suite is the second and last place that may import it.
|
||||||
|
ARCHIVED_SOURCES = Path(__file__).resolve().parents[2] / "legacy_cli_archive" / "src"
|
||||||
|
|
||||||
EXIFTOOL = shutil.which("exiftool")
|
EXIFTOOL = shutil.which("exiftool")
|
||||||
pytestmark = pytest.mark.skipif(EXIFTOOL is None, reason="exiftool not installed")
|
pytestmark = pytest.mark.skipif(EXIFTOOL is None, reason="exiftool not installed")
|
||||||
|
|
||||||
@@ -31,6 +37,7 @@ def _tag(path, keyword):
|
|||||||
|
|
||||||
|
|
||||||
def test_extracted_marks_and_partition_match_donor(tmp_path):
|
def test_extracted_marks_and_partition_match_donor(tmp_path):
|
||||||
|
sys.path.insert(0, str(ARCHIVED_SOURCES)) # the donor lives in the archive now
|
||||||
donor_exif = pytest.importorskip("nsfwtag.exif")
|
donor_exif = pytest.importorskip("nsfwtag.exif")
|
||||||
|
|
||||||
nsfw = _jpeg(tmp_path / "nsfw.jpg", 1)
|
nsfw = _jpeg(tmp_path / "nsfw.jpg", 1)
|
||||||
|
|||||||
@@ -143,6 +143,10 @@
|
|||||||
],
|
],
|
||||||
"US06-06": [
|
"US06-06": [
|
||||||
"tests/e2e/test_phase_f_pipeline.py"
|
"tests/e2e/test_phase_f_pipeline.py"
|
||||||
|
],
|
||||||
|
"US07-01": [
|
||||||
|
"tests/unit/test_legacy_archive.py",
|
||||||
|
"tests/integration/test_legacy_import.py"
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
173
tests/unit/test_legacy_archive.py
Normal file
173
tests/unit/test_legacy_archive.py
Normal file
@@ -0,0 +1,173 @@
|
|||||||
|
"""Archive lint (US07-01): the frozen CLI archive is complete, honest, and inert.
|
||||||
|
|
||||||
|
Freezing the donors is only worth something if three things stay true: the archive
|
||||||
|
still holds exactly the bytes it claims, it holds no secret, and production cannot
|
||||||
|
reach it. Each of those is a one-line mistake away — a helpful `sys.path` insert, a
|
||||||
|
copied `.env`, an edited "just this once" source — so each is asserted here.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
import yaml
|
||||||
|
|
||||||
|
REPO = Path(__file__).resolve().parents[2]
|
||||||
|
ARCHIVE = REPO / "legacy_cli_archive"
|
||||||
|
SOURCES = ARCHIVE / "src"
|
||||||
|
LEDGER = ARCHIVE / "donor_ledger.yaml"
|
||||||
|
CHECKSUMS = ARCHIVE / "CHECKSUMS.sha256"
|
||||||
|
|
||||||
|
# Everything the concept requires an archive to carry (§3 "Donor-first CLI
|
||||||
|
# migration and archival"): the sources, their docs, a dependency lock, schema
|
||||||
|
# notes, a redacted sample configuration, the ledger, and recorded checksums.
|
||||||
|
REQUIRED_ARTIFACTS = (
|
||||||
|
"README.md",
|
||||||
|
"donor_ledger.yaml",
|
||||||
|
"CHECKSUMS.sha256",
|
||||||
|
"requirements-lock.txt",
|
||||||
|
"photo_analyzer.env.sample",
|
||||||
|
)
|
||||||
|
ARCHIVED_MODULES = ("photo_analyzer", "nsfwtag", "webapp", "nsfw_tag", "compare_models")
|
||||||
|
# Only the two suites that compare against the donors may put the archive on the
|
||||||
|
# import path; every other tree must not name it at all.
|
||||||
|
IMPORT_ALLOWED = {
|
||||||
|
REPO / "tests" / "characterization" / "conftest.py",
|
||||||
|
REPO / "tests" / "characterization" / "test_donor_ledger.py", # lints the ledger there
|
||||||
|
REPO / "tests" / "integration" / "test_safety_parity.py",
|
||||||
|
Path(__file__),
|
||||||
|
}
|
||||||
|
SECRET_PATTERNS = (
|
||||||
|
re.compile(r"sk-(?!REPLACE_WITH_YOUR_KEY)[A-Za-z0-9_\-]{16,}"),
|
||||||
|
re.compile(r"AIza[0-9A-Za-z_\-]{20,}"), # Google API keys
|
||||||
|
re.compile(r"(?i)api[_-]?key\s*[=:]\s*['\"][A-Za-z0-9_\-]{16,}['\"]"),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _archived_files() -> list[Path]:
|
||||||
|
return sorted(p for p in SOURCES.rglob("*") if p.is_file() and "__pycache__" not in p.parts)
|
||||||
|
|
||||||
|
|
||||||
|
# ── completeness ─────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_archive_carries_every_required_artifact():
|
||||||
|
for artifact in REQUIRED_ARTIFACTS:
|
||||||
|
path = ARCHIVE / artifact
|
||||||
|
assert path.is_file(), f"archive is missing {artifact}"
|
||||||
|
assert path.stat().st_size > 0, f"archive artifact {artifact} is empty"
|
||||||
|
assert (SOURCES / "photo_analyzer.py").is_file()
|
||||||
|
assert (SOURCES / "nsfwtag" / "README.md").is_file(), "donor docs must be archived too"
|
||||||
|
assert (SOURCES / "webapp" / "README.md").is_file()
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_ledger_source_is_present_in_the_archive():
|
||||||
|
rows = yaml.safe_load(LEDGER.read_text(encoding="utf-8"))["rows"]
|
||||||
|
for row in rows:
|
||||||
|
source = SOURCES / row["source"]["file"]
|
||||||
|
assert source.is_file(), f"{row['id']}: {row['source']['file']} is not archived"
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_readme_records_provenance_and_the_no_import_rule():
|
||||||
|
readme = (ARCHIVE / "README.md").read_text(encoding="utf-8")
|
||||||
|
for expected in ("CHECKSUMS.sha256", "requirements-lock.txt", "donor_ledger.yaml"):
|
||||||
|
assert expected in readme, f"README does not point at {expected}"
|
||||||
|
assert "Schema notes" in readme, "the donor's schema must be documented"
|
||||||
|
assert "nsfw_scores.csv" in readme, "the CSV's fate must be documented"
|
||||||
|
|
||||||
|
|
||||||
|
# ── integrity ────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_archived_source_matches_its_checksum():
|
||||||
|
"""A frozen archive that silently drifts is not evidence of anything."""
|
||||||
|
recorded = {}
|
||||||
|
for line in CHECKSUMS.read_text(encoding="utf-8").splitlines():
|
||||||
|
digest, _, rel = line.partition(" ")
|
||||||
|
if rel:
|
||||||
|
recorded[rel.strip()] = digest
|
||||||
|
actual = {
|
||||||
|
str(path.relative_to(ARCHIVE)): hashlib.sha256(path.read_bytes()).hexdigest()
|
||||||
|
for path in _archived_files()
|
||||||
|
}
|
||||||
|
assert actual == recorded, "archived sources and CHECKSUMS.sha256 disagree"
|
||||||
|
|
||||||
|
|
||||||
|
# ── redaction ────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_archive_contains_no_credential():
|
||||||
|
for path in [*_archived_files(), *(ARCHIVE / a for a in REQUIRED_ARTIFACTS)]:
|
||||||
|
if path.suffix in (".png", ".jpg", ".webp"):
|
||||||
|
continue
|
||||||
|
text = path.read_text(encoding="utf-8", errors="replace")
|
||||||
|
for pattern in SECRET_PATTERNS:
|
||||||
|
assert not pattern.search(text), f"possible secret in {path.relative_to(REPO)}"
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_sample_configuration_is_a_placeholder_only():
|
||||||
|
sample = (ARCHIVE / "photo_analyzer.env.sample").read_text(encoding="utf-8")
|
||||||
|
assert "REDACTED" in sample
|
||||||
|
# Placeholders are not key-shaped, so neither a scanner nor a reader can
|
||||||
|
# mistake the sample for a credential.
|
||||||
|
assert "LLM_API_KEY=<" in sample
|
||||||
|
# No real env file, database, log, or CSV may ride along in the archive.
|
||||||
|
strays = [
|
||||||
|
p.name
|
||||||
|
for p in ARCHIVE.rglob("*")
|
||||||
|
if p.is_file() and p.suffix in (".env", ".db", ".sqlite", ".sqlite3", ".log", ".csv")
|
||||||
|
]
|
||||||
|
assert strays == [], f"unexpected runtime files archived: {strays}"
|
||||||
|
|
||||||
|
|
||||||
|
# ── inertness ────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_production_code_never_imports_an_archived_module():
|
||||||
|
for path in (REPO / "photo_pipeline").rglob("*.py"):
|
||||||
|
text = path.read_text(encoding="utf-8")
|
||||||
|
for module in ARCHIVED_MODULES:
|
||||||
|
assert not re.search(rf"^\s*(import|from)\s+{module}\b", text, re.MULTILINE), \
|
||||||
|
f"{path.relative_to(REPO)} imports the archived {module}"
|
||||||
|
assert "legacy_cli_archive" not in text, \
|
||||||
|
f"{path.relative_to(REPO)} names the archive"
|
||||||
|
|
||||||
|
|
||||||
|
def test_only_the_parity_suites_put_the_archive_on_the_import_path():
|
||||||
|
for path in (REPO / "tests").rglob("*.py"):
|
||||||
|
if path in IMPORT_ALLOWED or "__pycache__" in path.parts:
|
||||||
|
continue
|
||||||
|
assert "legacy_cli_archive" not in path.read_text(encoding="utf-8"), \
|
||||||
|
f"{path.relative_to(REPO)} reaches into the archive"
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_archived_modules_are_unimportable_from_a_clean_interpreter():
|
||||||
|
"""The real check: a fresh process with the repo on its path cannot load them."""
|
||||||
|
script = (
|
||||||
|
"import importlib.util, sys; "
|
||||||
|
f"sys.path.insert(0, {str(REPO)!r}); "
|
||||||
|
"print([m for m in "
|
||||||
|
f"{list(ARCHIVED_MODULES)!r}"
|
||||||
|
" if importlib.util.find_spec(m) is not None])"
|
||||||
|
)
|
||||||
|
result = subprocess.run(
|
||||||
|
[sys.executable, "-c", script], capture_output=True, text=True, cwd=str(REPO), check=True
|
||||||
|
)
|
||||||
|
assert result.stdout.strip() == "[]", f"still importable: {result.stdout.strip()}"
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_application_starts_without_the_archive(tmp_path):
|
||||||
|
"""Nothing in the runtime path may need the frozen sources to exist."""
|
||||||
|
pytest.importorskip("fastapi")
|
||||||
|
from photo_pipeline.api.app import create_app
|
||||||
|
from photo_pipeline.config import Config
|
||||||
|
|
||||||
|
config = Config.from_env(
|
||||||
|
{"PHOTO_PIPELINE_DATA_DIR": str(tmp_path / "data"), "PHOTO_PIPELINE_LIBRARY_ROOTS": ""}
|
||||||
|
)
|
||||||
|
assert create_app(config) is not None
|
||||||
Reference in New Issue
Block a user