V4.7 Phase 1: ingest orientation normalization, ProcessingArtifact removal

Move orientation normalization to the Source-ingest boundary and delete the
ProcessingArtifact subsystem it was built to serve.

Stored pages are now already upright, so nothing downstream derives a rotated
copy: every stored byte is the byte a provider is later sent. Rotation runs in
store_source_file ahead of hashing, so source.file_hash and file_size_bytes
describe exactly what is on disk. normalize_orientation becomes bytes-in /
bytes-out, and JPEG output reuses the source quantization tables and chroma
subsampling instead of re-quantizing at a fixed quality - measured at 50.3-56.1
dB PSNR at -6% size, against 50.0-53.5 dB at +38% for quality=95.

ProcessingArtifact held 2 rows against 77 successful transcriptions; the
subsystem effectively never ran. Deleting it removes the artifact cluster from
sources.py, the derivative resolution in workflows.py, the pre-provider commit
that only existed to make an artifact row durable, and the artifact evidence
dump from the Source detail page. The transcription_quality_warnings payload
folds into execution_attempt.normalized_metadata, so that feature keeps working
without the table.

tools/migrate_v46_to_v47.py carries steps 1 and 2: it rotated the 58 stored
images carrying EXIF orientation 3 in place, updated their recorded hash and
size, dropped processing_artifact and removed its one external file. It is
idempotent, keyed on state rather than a version marker.

tools/migrate_v45_to_v46.py is deleted. That migration is complete, and after
V4.7 it would restore a V4.5 backup into a schema that no longer matches.

Also fixes tests/test_config.py, which read the developer's local .env and
failed whenever WORKER_MAX_RETRIES was set.

Co-authored-by: Copilot App <[email protected]>
This commit is contained in:
zoltan57
2026-08-18 10:16:38 -05:00
co-authored by Copilot App
parent 246d7f9434
commit f86c0ff27b
15 changed files with 494 additions and 1240 deletions
-329
View File
@@ -1,329 +0,0 @@
"""One-time migration of a V4.5 database into the re-leveled V4.6 schema.
V4.6 re-levels the schema from the current SQLModel metadata rather than
running a chain of hand-rolled upgrade functions. The column sets are
unchanged; what changed is index coverage ([HIGH-04]), the ``use_alter``
break in the ``source``/``execution_attempt`` foreign key cycle, and the
relationship loading strategy ([CRIT-02]). This script therefore performs a
faithful, foreign-key-ordered row copy.
Design notes:
- The backup is read with plain ``sqlite3`` rather than through the ORM. The
V4.5 file is not guaranteed to satisfy the V4.6 mappers, and reading raw
rows means no relationship is ever traversed, so ``lazy="raise"`` cannot
bite.
- The target is written through SQLAlchemy Core against the live metadata, so
the same script works against PostgreSQL when that cutover happens.
- Identity is preserved exactly: UUIDs, digests, timestamps, attempt numbers,
and ``preferred_execution_attempt_id`` selections carry across unchanged.
No evidence payload is reinterpreted, normalized, or regenerated.
- No on-disk Source file, portrait, or artifact file is read for writing or
modified. ``--verify-artifacts`` reads artifact files, but only to hash
them.
- The script is idempotent: a row whose primary key already exists in the
target is skipped, never updated. It is never invoked from application
startup and never runs in the test suite.
Usage::
python tools/migrate_v45_to_v46.py --dry-run
python tools/migrate_v45_to_v46.py --verify-artifacts
"""
from __future__ import annotations
import argparse
import json
import sqlite3
import sys
from collections.abc import Iterator
from collections.abc import Sequence
from datetime import date
from datetime import datetime
from pathlib import Path
from typing import Any
from uuid import UUID
from sqlalchemy import Column
from sqlalchemy import Table
from sqlalchemy import create_engine
from sqlalchemy import insert
from sqlalchemy import inspect as sqlalchemy_inspect
from sqlalchemy import select
from sqlalchemy import update
from sqlalchemy.engine import Connection
from sqlmodel import SQLModel
from transcription.config import Settings
from transcription.config import get_settings
from transcription.db import models as _models # noqa: F401 (registers every table)
from transcription.db.engine import get_database_url
DEFAULT_BACKUP = Path("data/transcription.db.pre-v46.bak")
#: ``source.preferred_execution_attempt_id`` points at ``execution_attempt``,
#: which points back at ``source``. The cycle is broken with ``use_alter`` in
#: the metadata, so ``source`` rows are inserted with the column cleared and
#: the selections are replayed once ``execution_attempt`` is populated.
DEFERRED_TABLE = "source"
DEFERRED_COLUMN = "preferred_execution_attempt_id"
#: Row counts the V4.5 backup is expected to carry, used as a pre-flight guard
#: so the script cannot silently run against the wrong file.
EXPECTED_SOURCE_COUNTS = {
"document": 8,
"document_person": 11,
"document_type": 7,
"execution_attempt": 80,
"job": 11,
"job_source": 79,
"person": 5,
"person_role": 3,
"processing_artifact": 2,
"source": 76,
}
def _coerce(column: Column[Any], value: object) -> object:
"""Convert a raw SQLite value into what the target column's type binds.
SQLite hands back strings and integers; the V4.6 columns bind ``UUID``,
``datetime``, ``date``, ``bool``, enum members, and decoded JSON. The
conversion is lossless in both directions.
"""
if value is None:
return None
match type(column.type).__name__:
case "Uuid":
return value if isinstance(value, UUID) else UUID(str(value))
case "DateTime":
return value if isinstance(value, datetime) else datetime.fromisoformat(str(value))
case "Date":
return value if isinstance(value, date) else date.fromisoformat(str(value))
case "Boolean":
return bool(value)
case "JSONBCompat":
if isinstance(value, str | bytes | bytearray):
return json.loads(value)
return value
case "Enum":
enum_class = getattr(column.type, "enum_class", None)
if enum_class is None:
return value
# The same JobSourceStatus enum is persisted by value on
# job_source.status and by name on execution_attempt.status,
# because only the former declares values_callable. Accept either
# spelling so the copy round-trips both columns faithfully.
try:
return enum_class(value)
except ValueError:
return enum_class[str(value)]
case _:
return value
def _read_table(backup: sqlite3.Connection, table: Table) -> list[dict[str, object]]:
"""Read every row of ``table`` from the backup, coerced for the target."""
names = [column.name for column in table.columns]
quoted = ", ".join(f'"{name}"' for name in names)
rows: list[dict[str, object]] = []
for raw in backup.execute(f'select {quoted} from "{table.name}"'):
rows.append({name: _coerce(table.columns[name], raw[index]) for index, name in enumerate(names)})
return rows
def _primary_key(table: Table) -> Column[Any]:
columns = list(table.primary_key.columns)
if len(columns) != 1:
message = f"{table.name} does not have a single-column primary key"
raise RuntimeError(message)
return columns[0]
def _existing_keys(connection: Connection, table: Table) -> set[object]:
key = _primary_key(table)
return set(connection.execute(select(key)).scalars().all())
def _chunked(rows: Sequence[dict[str, object]], size: int = 200) -> Iterator[Sequence[dict[str, object]]]:
for start in range(0, len(rows), size):
yield rows[start : start + size]
def _verify_artifacts(settings: Settings) -> int:
"""Re-hash every migrated artifact through the service's own verifier."""
from transcription.db.models import ProcessingArtifact
from transcription.services.sources import SourceService
engine = create_engine(_sync_url(settings))
with engine.connect() as connection:
rows = connection.execute(select(SQLModel.metadata.tables["processing_artifact"])).mappings().all()
engine.dispose()
service = SourceService(settings=settings)
artifacts = [ProcessingArtifact(**dict(row)) for row in rows]
# Reuses the application's own integrity check so the migration cannot
# disagree with what the running app considers a valid artifact.
service._verify_artifacts_integrity(artifacts)
return len(artifacts)
def _sync_url(settings: Settings) -> str:
"""Return the target database URL with any async driver stripped."""
url = get_database_url(settings)
return url.replace("+aiosqlite", "").replace("+asyncpg", "").replace("+psycopg", "")
def _preflight(backup: sqlite3.Connection, *, strict: bool) -> None:
actual = {
name: backup.execute(f'select count(*) from "{name}"').fetchone()[0]
for name in EXPECTED_SOURCE_COUNTS
}
mismatched = {
name: (count, EXPECTED_SOURCE_COUNTS[name])
for name, count in actual.items()
if count != EXPECTED_SOURCE_COUNTS[name]
}
if not mismatched:
return
detail = ", ".join(f"{name}: found {found}, expected {want}" for name, (found, want) in sorted(mismatched.items()))
message = f"Backup row counts do not match the recorded V4.5 snapshot ({detail})"
if strict:
raise RuntimeError(message)
print(f"WARNING: {message}", file=sys.stderr)
def _copy_tables(
connection: Connection,
payload: dict[str, list[dict[str, object]]],
*,
dry_run: bool,
) -> tuple[int, dict[object, object]]:
"""Insert every missing row, deferring the cyclic foreign key column."""
deferred: dict[object, object] = {}
inserted_total = 0
for table in SQLModel.metadata.sorted_tables:
rows = payload[table.name]
existing = set() if dry_run else _existing_keys(connection, table)
key_name = _primary_key(table).name
pending = [row for row in rows if row[key_name] not in existing]
if table.name == DEFERRED_TABLE:
for row in pending:
selection = row[DEFERRED_COLUMN]
if selection is not None:
deferred[row[key_name]] = selection
row[DEFERRED_COLUMN] = None
if pending and not dry_run:
for chunk in _chunked(pending):
connection.execute(insert(table), list(chunk))
inserted_total += len(pending)
print(f" {table.name:24} insert={len(pending):<5} skip={len(rows) - len(pending)}")
return inserted_total, deferred
def _replay_deferred(connection: Connection, deferred: dict[object, object], *, dry_run: bool) -> None:
"""Restore the preferred-attempt selections held back by the FK cycle."""
if not deferred:
return
print(f" replaying {len(deferred)} deferred {DEFERRED_TABLE}.{DEFERRED_COLUMN} selection(s)")
if dry_run:
return
source = SQLModel.metadata.tables[DEFERRED_TABLE]
key = _primary_key(source)
for source_id, attempt_id in deferred.items():
connection.execute(update(source).where(key == source_id).values({DEFERRED_COLUMN: attempt_id}))
def _report_counts(connection: Connection) -> None:
print("\nPost-migration row counts:")
for table in SQLModel.metadata.sorted_tables:
actual = len(connection.execute(select(_primary_key(table))).all())
expected = EXPECTED_SOURCE_COUNTS.get(table.name)
flag = "" if expected is None or actual == expected else f" <-- expected {expected}"
print(f" {table.name:24} {actual}{flag}")
def _load_payload(backup_path: Path, *, strict_counts: bool) -> dict[str, list[dict[str, object]]]:
if not backup_path.is_file():
message = f"Backup database not found: {backup_path}"
raise FileNotFoundError(message)
backup = sqlite3.connect(f"file:{backup_path}?mode=ro", uri=True)
try:
_preflight(backup, strict=strict_counts)
return {table.name: _read_table(backup, table) for table in SQLModel.metadata.sorted_tables}
finally:
backup.close()
def migrate(*, backup_path: Path, settings: Settings, dry_run: bool, strict_counts: bool) -> int:
"""Copy every row from the V4.5 backup into the re-leveled schema."""
payload = _load_payload(backup_path, strict_counts=strict_counts)
engine = create_engine(_sync_url(settings))
try:
if not sqlalchemy_inspect(engine).has_table("document"):
print("Target schema is empty; creating it from the current metadata.")
if not dry_run:
SQLModel.metadata.create_all(engine)
with engine.begin() as connection:
inserted_total, deferred = _copy_tables(connection, payload, dry_run=dry_run)
_replay_deferred(connection, deferred, dry_run=dry_run)
if dry_run:
print("\nDry run: no rows were written.")
return inserted_total
with engine.connect() as connection:
_report_counts(connection)
finally:
engine.dispose()
return inserted_total
def main(argv: Sequence[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--backup", type=Path, default=DEFAULT_BACKUP, help="V4.5 database to read from")
parser.add_argument("--dry-run", action="store_true", help="Report what would be copied without writing")
parser.add_argument(
"--allow-count-mismatch",
action="store_true",
help="Warn instead of aborting when the backup row counts differ from the recorded snapshot",
)
parser.add_argument(
"--verify-artifacts",
action="store_true",
help="Re-hash every migrated processing artifact after the copy",
)
args = parser.parse_args(argv)
settings = get_settings()
print(f"Source: {args.backup}")
print(f"Target: {_sync_url(settings)}\n")
inserted = migrate(
backup_path=args.backup,
settings=settings,
dry_run=args.dry_run,
strict_counts=not args.allow_count_mismatch,
)
if args.verify_artifacts and not args.dry_run:
verified = _verify_artifacts(settings)
print(f"\nArtifact integrity verified for {verified} artifact(s).")
print(f"\nDone. {inserted} row(s) inserted.")
return 0
if __name__ == "__main__":
raise SystemExit(main())
+248
View File
@@ -0,0 +1,248 @@
"""One-time migration of a V4.6 database into the V4.7 schema.
V4.7 is an architectural cleanup: no new user-facing behaviour, but three
structural changes plus a one-time image backfill. This tool carries all of
them, and is built up phase by phase so the live database stays usable at
every phase boundary.
Steps, in execution order:
1. Rotate every stored Source image that still carries a supported EXIF
orientation, in place, and update ``source.file_hash`` and
``source.file_size_bytes`` to describe the rewritten file.
2. Drop the ``processing_artifact`` table and delete its external files.
Design notes:
- The image rewrite reuses the application's own
:func:`~transcription.services.normalization.normalize_orientation`, so the
backfilled bytes are byte-identical to what ingest would now produce. It
reuses the source quantization tables and subsampling rather than
re-quantizing, which is both smaller and higher fidelity than a fixed
quality setting.
- The hash and size are rewritten alongside the file. After V4.7 the evidence
digest is derived straight from ``source.file_hash``, so leaving it
describing the pre-rotation bytes would silently invalidate every future
export.
- The database is read and written through SQLAlchemy Core against the live
metadata, so the same script works against PostgreSQL when that cutover
happens. Raw DDL is used only for the table drop, which has no Core
equivalent that is safe to express against deleted metadata.
- The script is idempotent, keyed on state rather than on a version marker:
an image with no supported orientation tag is skipped, and a table that is
already absent is skipped. It is never invoked from application startup and
never runs in the test suite.
- **The application must not be running.** The image rewrite is not atomic
with the row update, and SQLite will refuse the schema change while another
connection holds the database.
Usage::
python tools/migrate_v46_to_v47.py --dry-run
python tools/migrate_v46_to_v47.py
"""
from __future__ import annotations
import argparse
import hashlib
import sys
from collections.abc import Sequence
from pathlib import Path
from sqlalchemy import create_engine
from sqlalchemy import inspect as sqlalchemy_inspect
from sqlalchemy import select
from sqlalchemy import text
from sqlalchemy import update
from sqlalchemy.engine import Connection
from sqlmodel import SQLModel
from transcription.config import Settings
from transcription.config import get_settings
from transcription.db import models as _models # noqa: F401 (registers every table)
from transcription.db.engine import get_database_url
from transcription.services.normalization import normalize_orientation
from transcription.services.sources import source_mime_type
#: Row counts the V4.6 database is expected to carry, used as a pre-flight
#: guard so the script cannot silently run against the wrong file.
EXPECTED_ROW_COUNTS = {
"document": 8,
"document_person": 11,
"document_type": 7,
"execution_attempt": 80,
"job": 11,
"job_source": 79,
"person": 5,
"person_role": 3,
"source": 76,
}
ARTIFACT_TABLE = "processing_artifact"
#: The V4.6 default for the deleted ``Settings.artifact_dir``. The setting no
#: longer exists, so the historical location is recorded here instead.
DEFAULT_ARTIFACT_DIR = Path("data/artifacts")
def _sync_url(settings: Settings) -> str:
"""Return the target database URL with any async driver stripped."""
url = get_database_url(settings)
return url.replace("+aiosqlite", "").replace("+asyncpg", "").replace("+psycopg", "")
def _preflight(connection: Connection, *, strict: bool) -> None:
inspector = sqlalchemy_inspect(connection)
present = set(inspector.get_table_names())
mismatched: dict[str, tuple[object, int]] = {}
for name, expected in EXPECTED_ROW_COUNTS.items():
if name not in present:
mismatched[name] = ("missing", expected)
continue
actual = connection.execute(text(f'select count(*) from "{name}"')).scalar_one()
if actual != expected:
mismatched[name] = (actual, expected)
if not mismatched:
return
detail = ", ".join(f"{name}: found {found}, expected {want}" for name, (found, want) in sorted(mismatched.items()))
message = f"Database row counts do not match the recorded V4.6 snapshot ({detail})"
if strict:
raise RuntimeError(message)
print(f"WARNING: {message}", file=sys.stderr)
def rotate_stored_images(connection: Connection, *, dry_run: bool) -> int:
"""Step 1: rewrite every mis-oriented stored image and its recorded digest."""
source = SQLModel.metadata.tables["source"]
rows = connection.execute(
select(source.c.id, source.c.file_path, source.c.filename)
).all()
rotated = 0
missing = 0
for source_id, file_path, filename in rows:
path = Path(str(file_path))
if not path.is_file():
print(f" WARNING: source file not found, skipped: {path}", file=sys.stderr)
missing += 1
continue
content = path.read_bytes()
normalized = normalize_orientation(content, media_type=source_mime_type(str(filename)))
if normalized is None:
continue
rotated += 1
print(
f" {path.name} orientation={normalized.original_orientation} "
f"rotation={normalized.applied_rotation_degrees} "
f"{len(content)} -> {len(normalized.content)} bytes"
)
if dry_run:
continue
path.write_bytes(normalized.content)
connection.execute(
update(source)
.where(source.c.id == source_id)
.values(
file_hash=hashlib.sha256(normalized.content).hexdigest(),
file_size_bytes=len(normalized.content),
)
)
print(f" rotated={rotated} upright={len(rows) - rotated - missing} missing={missing}")
return rotated
def drop_processing_artifacts(connection: Connection, artifact_dir: Path, *, dry_run: bool) -> int:
"""Step 2: drop the artifact table and delete the files it referenced."""
inspector = sqlalchemy_inspect(connection)
if ARTIFACT_TABLE not in set(inspector.get_table_names()):
print(f" {ARTIFACT_TABLE} already absent")
return 0
references = [
str(row[0])
for row in connection.execute(
text(f'select external_reference from "{ARTIFACT_TABLE}" where external_reference is not null')
)
]
count = connection.execute(text(f'select count(*) from "{ARTIFACT_TABLE}"')).scalar_one()
print(f" dropping {ARTIFACT_TABLE} ({count} row(s), {len(references)} external file(s))")
if dry_run:
return count
connection.execute(text(f'drop table "{ARTIFACT_TABLE}"'))
artifact_root = artifact_dir.resolve()
for reference in references:
relative = Path(reference)
if relative.is_absolute() or ".." in relative.parts:
print(f" WARNING: skipped unsafe artifact reference: {reference}", file=sys.stderr)
continue
artifact_path = (artifact_root / relative).resolve()
if artifact_root not in artifact_path.parents:
print(f" WARNING: skipped artifact outside root: {reference}", file=sys.stderr)
continue
artifact_path.unlink(missing_ok=True)
parent = artifact_path.parent
if parent != artifact_root and parent.is_dir() and not any(parent.iterdir()):
parent.rmdir()
return count
def migrate(*, settings: Settings, artifact_dir: Path, dry_run: bool, strict_counts: bool) -> None:
"""Apply every V4.7 migration step in order."""
engine = create_engine(_sync_url(settings))
try:
with engine.begin() as connection:
_preflight(connection, strict=strict_counts)
print("\nStep 1: rotate stored images")
rotate_stored_images(connection, dry_run=dry_run)
print(f"\nStep 2: drop {ARTIFACT_TABLE}")
drop_processing_artifacts(connection, artifact_dir, dry_run=dry_run)
finally:
engine.dispose()
if dry_run:
print("\nDry run: nothing was written.")
else:
print("\nDone.")
def main(argv: Sequence[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--dry-run", action="store_true", help="Report what would change without writing")
parser.add_argument(
"--allow-count-mismatch",
action="store_true",
help="Warn instead of aborting when row counts differ from the recorded V4.6 snapshot",
)
parser.add_argument(
"--artifact-dir",
type=Path,
default=DEFAULT_ARTIFACT_DIR,
help="Directory that held external artifact files before V4.7",
)
args = parser.parse_args(argv)
settings = get_settings()
print(f"Target: {_sync_url(settings)}")
migrate(
settings=settings,
artifact_dir=args.artifact_dir,
dry_run=args.dry_run,
strict_counts=not args.allow_count_mismatch,
)
return 0
if __name__ == "__main__":
raise SystemExit(main())