generated from john/python-template
V4.7 Phase 1: ingest orientation normalization, ProcessingArtifact removal
Move orientation normalization to the Source-ingest boundary and delete the ProcessingArtifact subsystem it was built to serve. Stored pages are now already upright, so nothing downstream derives a rotated copy: every stored byte is the byte a provider is later sent. Rotation runs in store_source_file ahead of hashing, so source.file_hash and file_size_bytes describe exactly what is on disk. normalize_orientation becomes bytes-in / bytes-out, and JPEG output reuses the source quantization tables and chroma subsampling instead of re-quantizing at a fixed quality - measured at 50.3-56.1 dB PSNR at -6% size, against 50.0-53.5 dB at +38% for quality=95. ProcessingArtifact held 2 rows against 77 successful transcriptions; the subsystem effectively never ran. Deleting it removes the artifact cluster from sources.py, the derivative resolution in workflows.py, the pre-provider commit that only existed to make an artifact row durable, and the artifact evidence dump from the Source detail page. The transcription_quality_warnings payload folds into execution_attempt.normalized_metadata, so that feature keeps working without the table. tools/migrate_v46_to_v47.py carries steps 1 and 2: it rotated the 58 stored images carrying EXIF orientation 3 in place, updated their recorded hash and size, dropped processing_artifact and removed its one external file. It is idempotent, keyed on state rather than a version marker. tools/migrate_v45_to_v46.py is deleted. That migration is complete, and after V4.7 it would restore a V4.5 backup into a schema that no longer matches. Also fixes tests/test_config.py, which read the developer's local .env and failed whenever WORKER_MAX_RETRIES was set. Co-authored-by: Copilot App <[email protected]>
This commit is contained in:
@@ -1,329 +0,0 @@
|
||||
"""One-time migration of a V4.5 database into the re-leveled V4.6 schema.
|
||||
|
||||
V4.6 re-levels the schema from the current SQLModel metadata rather than
|
||||
running a chain of hand-rolled upgrade functions. The column sets are
|
||||
unchanged; what changed is index coverage ([HIGH-04]), the ``use_alter``
|
||||
break in the ``source``/``execution_attempt`` foreign key cycle, and the
|
||||
relationship loading strategy ([CRIT-02]). This script therefore performs a
|
||||
faithful, foreign-key-ordered row copy.
|
||||
|
||||
Design notes:
|
||||
|
||||
- The backup is read with plain ``sqlite3`` rather than through the ORM. The
|
||||
V4.5 file is not guaranteed to satisfy the V4.6 mappers, and reading raw
|
||||
rows means no relationship is ever traversed, so ``lazy="raise"`` cannot
|
||||
bite.
|
||||
- The target is written through SQLAlchemy Core against the live metadata, so
|
||||
the same script works against PostgreSQL when that cutover happens.
|
||||
- Identity is preserved exactly: UUIDs, digests, timestamps, attempt numbers,
|
||||
and ``preferred_execution_attempt_id`` selections carry across unchanged.
|
||||
No evidence payload is reinterpreted, normalized, or regenerated.
|
||||
- No on-disk Source file, portrait, or artifact file is read for writing or
|
||||
modified. ``--verify-artifacts`` reads artifact files, but only to hash
|
||||
them.
|
||||
- The script is idempotent: a row whose primary key already exists in the
|
||||
target is skipped, never updated. It is never invoked from application
|
||||
startup and never runs in the test suite.
|
||||
|
||||
Usage::
|
||||
|
||||
python tools/migrate_v45_to_v46.py --dry-run
|
||||
python tools/migrate_v45_to_v46.py --verify-artifacts
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sqlite3
|
||||
import sys
|
||||
from collections.abc import Iterator
|
||||
from collections.abc import Sequence
|
||||
from datetime import date
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from uuid import UUID
|
||||
|
||||
from sqlalchemy import Column
|
||||
from sqlalchemy import Table
|
||||
from sqlalchemy import create_engine
|
||||
from sqlalchemy import insert
|
||||
from sqlalchemy import inspect as sqlalchemy_inspect
|
||||
from sqlalchemy import select
|
||||
from sqlalchemy import update
|
||||
from sqlalchemy.engine import Connection
|
||||
from sqlmodel import SQLModel
|
||||
|
||||
from transcription.config import Settings
|
||||
from transcription.config import get_settings
|
||||
from transcription.db import models as _models # noqa: F401 (registers every table)
|
||||
from transcription.db.engine import get_database_url
|
||||
|
||||
DEFAULT_BACKUP = Path("data/transcription.db.pre-v46.bak")
|
||||
|
||||
#: ``source.preferred_execution_attempt_id`` points at ``execution_attempt``,
|
||||
#: which points back at ``source``. The cycle is broken with ``use_alter`` in
|
||||
#: the metadata, so ``source`` rows are inserted with the column cleared and
|
||||
#: the selections are replayed once ``execution_attempt`` is populated.
|
||||
DEFERRED_TABLE = "source"
|
||||
DEFERRED_COLUMN = "preferred_execution_attempt_id"
|
||||
|
||||
#: Row counts the V4.5 backup is expected to carry, used as a pre-flight guard
|
||||
#: so the script cannot silently run against the wrong file.
|
||||
EXPECTED_SOURCE_COUNTS = {
|
||||
"document": 8,
|
||||
"document_person": 11,
|
||||
"document_type": 7,
|
||||
"execution_attempt": 80,
|
||||
"job": 11,
|
||||
"job_source": 79,
|
||||
"person": 5,
|
||||
"person_role": 3,
|
||||
"processing_artifact": 2,
|
||||
"source": 76,
|
||||
}
|
||||
|
||||
|
||||
def _coerce(column: Column[Any], value: object) -> object:
|
||||
"""Convert a raw SQLite value into what the target column's type binds.
|
||||
|
||||
SQLite hands back strings and integers; the V4.6 columns bind ``UUID``,
|
||||
``datetime``, ``date``, ``bool``, enum members, and decoded JSON. The
|
||||
conversion is lossless in both directions.
|
||||
"""
|
||||
if value is None:
|
||||
return None
|
||||
|
||||
match type(column.type).__name__:
|
||||
case "Uuid":
|
||||
return value if isinstance(value, UUID) else UUID(str(value))
|
||||
case "DateTime":
|
||||
return value if isinstance(value, datetime) else datetime.fromisoformat(str(value))
|
||||
case "Date":
|
||||
return value if isinstance(value, date) else date.fromisoformat(str(value))
|
||||
case "Boolean":
|
||||
return bool(value)
|
||||
case "JSONBCompat":
|
||||
if isinstance(value, str | bytes | bytearray):
|
||||
return json.loads(value)
|
||||
return value
|
||||
case "Enum":
|
||||
enum_class = getattr(column.type, "enum_class", None)
|
||||
if enum_class is None:
|
||||
return value
|
||||
# The same JobSourceStatus enum is persisted by value on
|
||||
# job_source.status and by name on execution_attempt.status,
|
||||
# because only the former declares values_callable. Accept either
|
||||
# spelling so the copy round-trips both columns faithfully.
|
||||
try:
|
||||
return enum_class(value)
|
||||
except ValueError:
|
||||
return enum_class[str(value)]
|
||||
case _:
|
||||
return value
|
||||
|
||||
|
||||
def _read_table(backup: sqlite3.Connection, table: Table) -> list[dict[str, object]]:
|
||||
"""Read every row of ``table`` from the backup, coerced for the target."""
|
||||
names = [column.name for column in table.columns]
|
||||
quoted = ", ".join(f'"{name}"' for name in names)
|
||||
rows: list[dict[str, object]] = []
|
||||
for raw in backup.execute(f'select {quoted} from "{table.name}"'):
|
||||
rows.append({name: _coerce(table.columns[name], raw[index]) for index, name in enumerate(names)})
|
||||
return rows
|
||||
|
||||
|
||||
def _primary_key(table: Table) -> Column[Any]:
|
||||
columns = list(table.primary_key.columns)
|
||||
if len(columns) != 1:
|
||||
message = f"{table.name} does not have a single-column primary key"
|
||||
raise RuntimeError(message)
|
||||
return columns[0]
|
||||
|
||||
|
||||
def _existing_keys(connection: Connection, table: Table) -> set[object]:
|
||||
key = _primary_key(table)
|
||||
return set(connection.execute(select(key)).scalars().all())
|
||||
|
||||
|
||||
def _chunked(rows: Sequence[dict[str, object]], size: int = 200) -> Iterator[Sequence[dict[str, object]]]:
|
||||
for start in range(0, len(rows), size):
|
||||
yield rows[start : start + size]
|
||||
|
||||
|
||||
def _verify_artifacts(settings: Settings) -> int:
|
||||
"""Re-hash every migrated artifact through the service's own verifier."""
|
||||
from transcription.db.models import ProcessingArtifact
|
||||
from transcription.services.sources import SourceService
|
||||
|
||||
engine = create_engine(_sync_url(settings))
|
||||
with engine.connect() as connection:
|
||||
rows = connection.execute(select(SQLModel.metadata.tables["processing_artifact"])).mappings().all()
|
||||
engine.dispose()
|
||||
|
||||
service = SourceService(settings=settings)
|
||||
artifacts = [ProcessingArtifact(**dict(row)) for row in rows]
|
||||
# Reuses the application's own integrity check so the migration cannot
|
||||
# disagree with what the running app considers a valid artifact.
|
||||
service._verify_artifacts_integrity(artifacts)
|
||||
return len(artifacts)
|
||||
|
||||
|
||||
def _sync_url(settings: Settings) -> str:
|
||||
"""Return the target database URL with any async driver stripped."""
|
||||
url = get_database_url(settings)
|
||||
return url.replace("+aiosqlite", "").replace("+asyncpg", "").replace("+psycopg", "")
|
||||
|
||||
|
||||
def _preflight(backup: sqlite3.Connection, *, strict: bool) -> None:
|
||||
actual = {
|
||||
name: backup.execute(f'select count(*) from "{name}"').fetchone()[0]
|
||||
for name in EXPECTED_SOURCE_COUNTS
|
||||
}
|
||||
mismatched = {
|
||||
name: (count, EXPECTED_SOURCE_COUNTS[name])
|
||||
for name, count in actual.items()
|
||||
if count != EXPECTED_SOURCE_COUNTS[name]
|
||||
}
|
||||
if not mismatched:
|
||||
return
|
||||
detail = ", ".join(f"{name}: found {found}, expected {want}" for name, (found, want) in sorted(mismatched.items()))
|
||||
message = f"Backup row counts do not match the recorded V4.5 snapshot ({detail})"
|
||||
if strict:
|
||||
raise RuntimeError(message)
|
||||
print(f"WARNING: {message}", file=sys.stderr)
|
||||
|
||||
|
||||
def _copy_tables(
|
||||
connection: Connection,
|
||||
payload: dict[str, list[dict[str, object]]],
|
||||
*,
|
||||
dry_run: bool,
|
||||
) -> tuple[int, dict[object, object]]:
|
||||
"""Insert every missing row, deferring the cyclic foreign key column."""
|
||||
deferred: dict[object, object] = {}
|
||||
inserted_total = 0
|
||||
|
||||
for table in SQLModel.metadata.sorted_tables:
|
||||
rows = payload[table.name]
|
||||
existing = set() if dry_run else _existing_keys(connection, table)
|
||||
key_name = _primary_key(table).name
|
||||
|
||||
pending = [row for row in rows if row[key_name] not in existing]
|
||||
|
||||
if table.name == DEFERRED_TABLE:
|
||||
for row in pending:
|
||||
selection = row[DEFERRED_COLUMN]
|
||||
if selection is not None:
|
||||
deferred[row[key_name]] = selection
|
||||
row[DEFERRED_COLUMN] = None
|
||||
|
||||
if pending and not dry_run:
|
||||
for chunk in _chunked(pending):
|
||||
connection.execute(insert(table), list(chunk))
|
||||
|
||||
inserted_total += len(pending)
|
||||
print(f" {table.name:24} insert={len(pending):<5} skip={len(rows) - len(pending)}")
|
||||
|
||||
return inserted_total, deferred
|
||||
|
||||
|
||||
def _replay_deferred(connection: Connection, deferred: dict[object, object], *, dry_run: bool) -> None:
|
||||
"""Restore the preferred-attempt selections held back by the FK cycle."""
|
||||
if not deferred:
|
||||
return
|
||||
print(f" replaying {len(deferred)} deferred {DEFERRED_TABLE}.{DEFERRED_COLUMN} selection(s)")
|
||||
if dry_run:
|
||||
return
|
||||
source = SQLModel.metadata.tables[DEFERRED_TABLE]
|
||||
key = _primary_key(source)
|
||||
for source_id, attempt_id in deferred.items():
|
||||
connection.execute(update(source).where(key == source_id).values({DEFERRED_COLUMN: attempt_id}))
|
||||
|
||||
|
||||
def _report_counts(connection: Connection) -> None:
|
||||
print("\nPost-migration row counts:")
|
||||
for table in SQLModel.metadata.sorted_tables:
|
||||
actual = len(connection.execute(select(_primary_key(table))).all())
|
||||
expected = EXPECTED_SOURCE_COUNTS.get(table.name)
|
||||
flag = "" if expected is None or actual == expected else f" <-- expected {expected}"
|
||||
print(f" {table.name:24} {actual}{flag}")
|
||||
|
||||
|
||||
def _load_payload(backup_path: Path, *, strict_counts: bool) -> dict[str, list[dict[str, object]]]:
|
||||
if not backup_path.is_file():
|
||||
message = f"Backup database not found: {backup_path}"
|
||||
raise FileNotFoundError(message)
|
||||
backup = sqlite3.connect(f"file:{backup_path}?mode=ro", uri=True)
|
||||
try:
|
||||
_preflight(backup, strict=strict_counts)
|
||||
return {table.name: _read_table(backup, table) for table in SQLModel.metadata.sorted_tables}
|
||||
finally:
|
||||
backup.close()
|
||||
|
||||
|
||||
def migrate(*, backup_path: Path, settings: Settings, dry_run: bool, strict_counts: bool) -> int:
|
||||
"""Copy every row from the V4.5 backup into the re-leveled schema."""
|
||||
payload = _load_payload(backup_path, strict_counts=strict_counts)
|
||||
|
||||
engine = create_engine(_sync_url(settings))
|
||||
try:
|
||||
if not sqlalchemy_inspect(engine).has_table("document"):
|
||||
print("Target schema is empty; creating it from the current metadata.")
|
||||
if not dry_run:
|
||||
SQLModel.metadata.create_all(engine)
|
||||
|
||||
with engine.begin() as connection:
|
||||
inserted_total, deferred = _copy_tables(connection, payload, dry_run=dry_run)
|
||||
_replay_deferred(connection, deferred, dry_run=dry_run)
|
||||
|
||||
if dry_run:
|
||||
print("\nDry run: no rows were written.")
|
||||
return inserted_total
|
||||
|
||||
with engine.connect() as connection:
|
||||
_report_counts(connection)
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
return inserted_total
|
||||
|
||||
|
||||
def main(argv: Sequence[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--backup", type=Path, default=DEFAULT_BACKUP, help="V4.5 database to read from")
|
||||
parser.add_argument("--dry-run", action="store_true", help="Report what would be copied without writing")
|
||||
parser.add_argument(
|
||||
"--allow-count-mismatch",
|
||||
action="store_true",
|
||||
help="Warn instead of aborting when the backup row counts differ from the recorded snapshot",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--verify-artifacts",
|
||||
action="store_true",
|
||||
help="Re-hash every migrated processing artifact after the copy",
|
||||
)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
settings = get_settings()
|
||||
print(f"Source: {args.backup}")
|
||||
print(f"Target: {_sync_url(settings)}\n")
|
||||
|
||||
inserted = migrate(
|
||||
backup_path=args.backup,
|
||||
settings=settings,
|
||||
dry_run=args.dry_run,
|
||||
strict_counts=not args.allow_count_mismatch,
|
||||
)
|
||||
|
||||
if args.verify_artifacts and not args.dry_run:
|
||||
verified = _verify_artifacts(settings)
|
||||
print(f"\nArtifact integrity verified for {verified} artifact(s).")
|
||||
|
||||
print(f"\nDone. {inserted} row(s) inserted.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,248 @@
|
||||
"""One-time migration of a V4.6 database into the V4.7 schema.
|
||||
|
||||
V4.7 is an architectural cleanup: no new user-facing behaviour, but three
|
||||
structural changes plus a one-time image backfill. This tool carries all of
|
||||
them, and is built up phase by phase so the live database stays usable at
|
||||
every phase boundary.
|
||||
|
||||
Steps, in execution order:
|
||||
|
||||
1. Rotate every stored Source image that still carries a supported EXIF
|
||||
orientation, in place, and update ``source.file_hash`` and
|
||||
``source.file_size_bytes`` to describe the rewritten file.
|
||||
2. Drop the ``processing_artifact`` table and delete its external files.
|
||||
|
||||
Design notes:
|
||||
|
||||
- The image rewrite reuses the application's own
|
||||
:func:`~transcription.services.normalization.normalize_orientation`, so the
|
||||
backfilled bytes are byte-identical to what ingest would now produce. It
|
||||
reuses the source quantization tables and subsampling rather than
|
||||
re-quantizing, which is both smaller and higher fidelity than a fixed
|
||||
quality setting.
|
||||
- The hash and size are rewritten alongside the file. After V4.7 the evidence
|
||||
digest is derived straight from ``source.file_hash``, so leaving it
|
||||
describing the pre-rotation bytes would silently invalidate every future
|
||||
export.
|
||||
- The database is read and written through SQLAlchemy Core against the live
|
||||
metadata, so the same script works against PostgreSQL when that cutover
|
||||
happens. Raw DDL is used only for the table drop, which has no Core
|
||||
equivalent that is safe to express against deleted metadata.
|
||||
- The script is idempotent, keyed on state rather than on a version marker:
|
||||
an image with no supported orientation tag is skipped, and a table that is
|
||||
already absent is skipped. It is never invoked from application startup and
|
||||
never runs in the test suite.
|
||||
- **The application must not be running.** The image rewrite is not atomic
|
||||
with the row update, and SQLite will refuse the schema change while another
|
||||
connection holds the database.
|
||||
|
||||
Usage::
|
||||
|
||||
python tools/migrate_v46_to_v47.py --dry-run
|
||||
python tools/migrate_v46_to_v47.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import sys
|
||||
from collections.abc import Sequence
|
||||
from pathlib import Path
|
||||
|
||||
from sqlalchemy import create_engine
|
||||
from sqlalchemy import inspect as sqlalchemy_inspect
|
||||
from sqlalchemy import select
|
||||
from sqlalchemy import text
|
||||
from sqlalchemy import update
|
||||
from sqlalchemy.engine import Connection
|
||||
from sqlmodel import SQLModel
|
||||
|
||||
from transcription.config import Settings
|
||||
from transcription.config import get_settings
|
||||
from transcription.db import models as _models # noqa: F401 (registers every table)
|
||||
from transcription.db.engine import get_database_url
|
||||
from transcription.services.normalization import normalize_orientation
|
||||
from transcription.services.sources import source_mime_type
|
||||
|
||||
#: Row counts the V4.6 database is expected to carry, used as a pre-flight
|
||||
#: guard so the script cannot silently run against the wrong file.
|
||||
EXPECTED_ROW_COUNTS = {
|
||||
"document": 8,
|
||||
"document_person": 11,
|
||||
"document_type": 7,
|
||||
"execution_attempt": 80,
|
||||
"job": 11,
|
||||
"job_source": 79,
|
||||
"person": 5,
|
||||
"person_role": 3,
|
||||
"source": 76,
|
||||
}
|
||||
|
||||
ARTIFACT_TABLE = "processing_artifact"
|
||||
|
||||
#: The V4.6 default for the deleted ``Settings.artifact_dir``. The setting no
|
||||
#: longer exists, so the historical location is recorded here instead.
|
||||
DEFAULT_ARTIFACT_DIR = Path("data/artifacts")
|
||||
|
||||
|
||||
def _sync_url(settings: Settings) -> str:
|
||||
"""Return the target database URL with any async driver stripped."""
|
||||
url = get_database_url(settings)
|
||||
return url.replace("+aiosqlite", "").replace("+asyncpg", "").replace("+psycopg", "")
|
||||
|
||||
|
||||
def _preflight(connection: Connection, *, strict: bool) -> None:
|
||||
inspector = sqlalchemy_inspect(connection)
|
||||
present = set(inspector.get_table_names())
|
||||
mismatched: dict[str, tuple[object, int]] = {}
|
||||
for name, expected in EXPECTED_ROW_COUNTS.items():
|
||||
if name not in present:
|
||||
mismatched[name] = ("missing", expected)
|
||||
continue
|
||||
actual = connection.execute(text(f'select count(*) from "{name}"')).scalar_one()
|
||||
if actual != expected:
|
||||
mismatched[name] = (actual, expected)
|
||||
if not mismatched:
|
||||
return
|
||||
detail = ", ".join(f"{name}: found {found}, expected {want}" for name, (found, want) in sorted(mismatched.items()))
|
||||
message = f"Database row counts do not match the recorded V4.6 snapshot ({detail})"
|
||||
if strict:
|
||||
raise RuntimeError(message)
|
||||
print(f"WARNING: {message}", file=sys.stderr)
|
||||
|
||||
|
||||
def rotate_stored_images(connection: Connection, *, dry_run: bool) -> int:
|
||||
"""Step 1: rewrite every mis-oriented stored image and its recorded digest."""
|
||||
source = SQLModel.metadata.tables["source"]
|
||||
rows = connection.execute(
|
||||
select(source.c.id, source.c.file_path, source.c.filename)
|
||||
).all()
|
||||
|
||||
rotated = 0
|
||||
missing = 0
|
||||
for source_id, file_path, filename in rows:
|
||||
path = Path(str(file_path))
|
||||
if not path.is_file():
|
||||
print(f" WARNING: source file not found, skipped: {path}", file=sys.stderr)
|
||||
missing += 1
|
||||
continue
|
||||
|
||||
content = path.read_bytes()
|
||||
normalized = normalize_orientation(content, media_type=source_mime_type(str(filename)))
|
||||
if normalized is None:
|
||||
continue
|
||||
|
||||
rotated += 1
|
||||
print(
|
||||
f" {path.name} orientation={normalized.original_orientation} "
|
||||
f"rotation={normalized.applied_rotation_degrees} "
|
||||
f"{len(content)} -> {len(normalized.content)} bytes"
|
||||
)
|
||||
if dry_run:
|
||||
continue
|
||||
|
||||
path.write_bytes(normalized.content)
|
||||
connection.execute(
|
||||
update(source)
|
||||
.where(source.c.id == source_id)
|
||||
.values(
|
||||
file_hash=hashlib.sha256(normalized.content).hexdigest(),
|
||||
file_size_bytes=len(normalized.content),
|
||||
)
|
||||
)
|
||||
|
||||
print(f" rotated={rotated} upright={len(rows) - rotated - missing} missing={missing}")
|
||||
return rotated
|
||||
|
||||
|
||||
def drop_processing_artifacts(connection: Connection, artifact_dir: Path, *, dry_run: bool) -> int:
|
||||
"""Step 2: drop the artifact table and delete the files it referenced."""
|
||||
inspector = sqlalchemy_inspect(connection)
|
||||
if ARTIFACT_TABLE not in set(inspector.get_table_names()):
|
||||
print(f" {ARTIFACT_TABLE} already absent")
|
||||
return 0
|
||||
|
||||
references = [
|
||||
str(row[0])
|
||||
for row in connection.execute(
|
||||
text(f'select external_reference from "{ARTIFACT_TABLE}" where external_reference is not null')
|
||||
)
|
||||
]
|
||||
count = connection.execute(text(f'select count(*) from "{ARTIFACT_TABLE}"')).scalar_one()
|
||||
print(f" dropping {ARTIFACT_TABLE} ({count} row(s), {len(references)} external file(s))")
|
||||
|
||||
if dry_run:
|
||||
return count
|
||||
|
||||
connection.execute(text(f'drop table "{ARTIFACT_TABLE}"'))
|
||||
|
||||
artifact_root = artifact_dir.resolve()
|
||||
for reference in references:
|
||||
relative = Path(reference)
|
||||
if relative.is_absolute() or ".." in relative.parts:
|
||||
print(f" WARNING: skipped unsafe artifact reference: {reference}", file=sys.stderr)
|
||||
continue
|
||||
artifact_path = (artifact_root / relative).resolve()
|
||||
if artifact_root not in artifact_path.parents:
|
||||
print(f" WARNING: skipped artifact outside root: {reference}", file=sys.stderr)
|
||||
continue
|
||||
artifact_path.unlink(missing_ok=True)
|
||||
parent = artifact_path.parent
|
||||
if parent != artifact_root and parent.is_dir() and not any(parent.iterdir()):
|
||||
parent.rmdir()
|
||||
|
||||
return count
|
||||
|
||||
|
||||
def migrate(*, settings: Settings, artifact_dir: Path, dry_run: bool, strict_counts: bool) -> None:
|
||||
"""Apply every V4.7 migration step in order."""
|
||||
engine = create_engine(_sync_url(settings))
|
||||
try:
|
||||
with engine.begin() as connection:
|
||||
_preflight(connection, strict=strict_counts)
|
||||
|
||||
print("\nStep 1: rotate stored images")
|
||||
rotate_stored_images(connection, dry_run=dry_run)
|
||||
|
||||
print(f"\nStep 2: drop {ARTIFACT_TABLE}")
|
||||
drop_processing_artifacts(connection, artifact_dir, dry_run=dry_run)
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
if dry_run:
|
||||
print("\nDry run: nothing was written.")
|
||||
else:
|
||||
print("\nDone.")
|
||||
|
||||
|
||||
def main(argv: Sequence[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--dry-run", action="store_true", help="Report what would change without writing")
|
||||
parser.add_argument(
|
||||
"--allow-count-mismatch",
|
||||
action="store_true",
|
||||
help="Warn instead of aborting when row counts differ from the recorded V4.6 snapshot",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--artifact-dir",
|
||||
type=Path,
|
||||
default=DEFAULT_ARTIFACT_DIR,
|
||||
help="Directory that held external artifact files before V4.7",
|
||||
)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
settings = get_settings()
|
||||
print(f"Target: {_sync_url(settings)}")
|
||||
|
||||
migrate(
|
||||
settings=settings,
|
||||
artifact_dir=args.artifact_dir,
|
||||
dry_run=args.dry_run,
|
||||
strict_counts=not args.allow_count_mismatch,
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user