generated from john/python-template
Removes the duplicated registry CRUD, the hand-written not-found raises, and the three divergent media writers. Behavior is preserved: every existing Document Type and Person Role test passes unchanged, which is the primary proof for MED-11. [MED-11] Generic registry service - New services/registry.py owns RegistryService[ModelT]: list, list with counts, create with IntegrityError -> conflict mapping, read, update, delete with built-in and referenced guards, is_referenced, and label normalization/casefold keying. - DocumentTypeRegistry and PersonRoleRegistry declare only the model, error class, noun, short noun, retainer phrase, and reference columns. - DocumentService and PeopleService keep their public method names and delegate. Every user-facing message, error category, and suggestion string is reproduced verbatim; only the noun is templated. - Deleted _normalize_registry_label, _document_type_label_key, _normalize_role_label, _person_role_label_key, _document_type_is_referenced, and _person_role_is_referenced. [MED-12] Shared not-found lookup - ServiceBase._get_or_raise(model, id, *, session, error, noun, suggestion, options) loads by primary key or raises the caller's error type. - documents.py: local _get_document_or_raise deleted; replaced by _read_document and adopted at read_document, delete_document, and set_document_type, which previously bypassed the helper and hand-wrote the raise. - sources.py: 8 identical Source raises and 1 Job raise collapsed into _read_source / _get_or_raise. - jobs.py and people.py already funneled through local _not_found builders and were left alone. [MED-13][MED-01] Single media writer - New services/media_storage.py owns validate -> name -> mkdir -> write -> wrap OSError. The write runs in asyncio.to_thread, so uploads no longer block the event loop. - store_source_file, store_person_portrait, and store_homepage_image now share it and are async. Callers in store.py, people_page.py, and home_page.py await them. mkdir failures are now also translated to a domain error instead of escaping as a raw OSError. - homepage_store gains HomepageStorageError so its write reports like the others. [MED-14, partial] Service independence - New services/source_media.py owns SOURCE_MIME_TYPES, SOURCE_EXTENSIONS, lookup_source_mime_type, and supported_source_formats. - documents.py no longer imports services/sources.py. Its print projection uses the non-raising lookup and raises DocumentError, so DocumentService no longer emits a TranscriptionError. - api/v4_print.py imports the mapping from the policy module. - store.py and workflows.py still import sources.py; both are orchestration modules, which services.instructions.md:75-77 explicitly permits. - Splitting SourceService itself remains deferred to V4.7. [LOW-08] Query shape - list_sources_detail filters job_id with a JOIN on JobSource instead of loading every Source and filtering in Python. - read_source_navigation replaces the full ordered-id scan and .index() with two row-value comparisons bounded by LIMIT 1. - list_processing_artifacts gains the limit parameter its summary sibling already had. - build_evidence_export runs artifact integrity hashing and file reads through asyncio.to_thread. Tests - tests/test_service_boundaries.py: AST guard asserting no service module imports a sibling service module, plus a guard that the scan is non-empty. - tests/services/test_transcription_service.py: asserts the job_id filter emits a JOIN, and that navigation emits exactly two LIMIT queries. - tests/services/test_store.py: the two storage tests are now async. Verified: 276 passed, 4 skipped; ruff check clean.
154 lines
5.7 KiB
Python
154 lines
5.7 KiB
Python
from pathlib import Path
|
|
from uuid import uuid4
|
|
|
|
import pytest
|
|
from sqlmodel import select
|
|
|
|
from transcription.config import Settings
|
|
from transcription.db.models import Document
|
|
from transcription.db.models import Job
|
|
from transcription.db.models import JobSource
|
|
from transcription.db.models import Source
|
|
from transcription.services.people import store_person_portrait
|
|
from transcription.services.sources import source_mime_type
|
|
from transcription.services.store import SourceStorageError
|
|
from transcription.services.store import create_document_job
|
|
from transcription.services.store import create_job_for_document
|
|
from transcription.services.store import store_source_file
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_create_job_for_document_requires_at_least_one_source(async_session, tmp_path):
|
|
document = Document(id=uuid4(), name="needs-upload")
|
|
async_session.add(document)
|
|
await async_session.commit()
|
|
|
|
settings = Settings(openrouter_api_key="test-key", upload_dir=tmp_path)
|
|
|
|
with pytest.raises(SourceStorageError):
|
|
await create_job_for_document(
|
|
document_id=document.id,
|
|
source_files=[],
|
|
session=async_session,
|
|
settings=settings,
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_create_job_for_document_sorts_sources_and_creates_links(async_session, tmp_path):
|
|
document = Document(id=uuid4(), name="ordered-upload-doc")
|
|
async_session.add(document)
|
|
await async_session.commit()
|
|
|
|
settings = Settings(openrouter_api_key="test-key", upload_dir=tmp_path)
|
|
|
|
result = await create_job_for_document(
|
|
document_id=document.id,
|
|
source_files=[
|
|
("folder/b_page.pdf", b"b"),
|
|
("folder/A_page.pdf", b"a"),
|
|
],
|
|
provider="openrouter",
|
|
model="test-model",
|
|
session=async_session,
|
|
settings=settings,
|
|
)
|
|
|
|
created_job = await async_session.get(Job, result.job_id)
|
|
assert created_job is not None
|
|
assert created_job.provider == "openrouter"
|
|
assert created_job.model == "test-model"
|
|
assert created_job.prompt_name == "transcribe_document.md"
|
|
assert created_job.user_prompt is not None
|
|
|
|
sources = (
|
|
await async_session.exec(
|
|
select(Source).where(Source.document_id == document.id).order_by(Source.page_number) # pyright: ignore[reportArgumentType]
|
|
)
|
|
).all()
|
|
assert [source.upload_name for source in sources] == ["A_page.pdf", "b_page.pdf"]
|
|
assert all(source.filename.endswith(".pdf") for source in sources)
|
|
assert all("A_page" not in source.filename and "b_page" not in source.filename for source in sources)
|
|
assert all(Path(source.filename).stem == str(source.id) for source in sources)
|
|
assert all(Path(source.file_path).parent == (tmp_path / "documents" / str(document.id)) for source in sources)
|
|
assert [source.file_hash for source in sources] == [
|
|
"ca978112ca1bbdcafac231b39a23dc4da786eff8147c4e72b9807785afee48bb",
|
|
"3e23e8160039594a33894f6564e1b1348bbd7a0088d42c4acb73eeaed59c009d",
|
|
]
|
|
assert [source.file_size_bytes for source in sources] == [1, 1]
|
|
|
|
job_sources = (await async_session.exec(select(JobSource).where(JobSource.job_id == result.job_id))).all()
|
|
assert len(job_sources) == 2
|
|
assert set(result.source_ids) == {job_source.source_id for job_source in job_sources}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_create_document_job_stores_source_under_document_id_directory(async_session, tmp_path):
|
|
settings = Settings(openrouter_api_key="test-key", upload_dir=tmp_path)
|
|
|
|
result = await create_document_job(
|
|
filename="single-page.jpg",
|
|
file_bytes=b"image-bytes",
|
|
session=async_session,
|
|
settings=settings,
|
|
)
|
|
|
|
expected_parent = tmp_path / "documents" / str(result.document_id)
|
|
assert result.stored_path.parent == expected_parent
|
|
assert result.stored_path.exists()
|
|
|
|
source = (
|
|
await async_session.exec(
|
|
select(Source).where(Source.document_id == result.document_id).order_by(Source.page_number) # pyright: ignore[reportArgumentType]
|
|
)
|
|
).first()
|
|
assert source is not None
|
|
assert Path(source.filename).stem == str(source.id)
|
|
assert result.stored_path.name == source.filename
|
|
assert Path(source.file_path).parent == expected_parent
|
|
assert source.file_hash == "2c8648d103e3dd7ad87660da0f126a1443b6d21ac1bd3ec000c5e24e2373a90c"
|
|
assert source.file_size_bytes == len(b"image-bytes")
|
|
|
|
created_job = await async_session.get(Job, result.job_id)
|
|
assert created_job is not None
|
|
assert created_job.prompt_name == "transcribe_document.md"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_store_person_portrait_stores_file_under_person_id_directory(tmp_path):
|
|
settings = Settings(openrouter_api_key="test-key", upload_dir=tmp_path)
|
|
person_id = uuid4()
|
|
|
|
stored_path = await store_person_portrait(
|
|
person_id=person_id,
|
|
filename="portrait.png",
|
|
file_bytes=b"portrait-bytes",
|
|
settings=settings,
|
|
)
|
|
|
|
assert stored_path.parent == (tmp_path / "persons" / str(person_id))
|
|
assert stored_path.exists()
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("filename", "expected_mime_type"),
|
|
[
|
|
("page.jpg", "image/jpeg"),
|
|
("page.JPEG", "image/jpeg"),
|
|
("page.png", "image/png"),
|
|
("page.tif", "image/tiff"),
|
|
("page.TIFF", "image/tiff"),
|
|
("page.pdf", "application/pdf"),
|
|
],
|
|
)
|
|
def test_source_mime_type_uses_canonical_source_policy(filename, expected_mime_type):
|
|
assert source_mime_type(filename) == expected_mime_type
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_source_storage_rejects_unsupported_format(tmp_path):
|
|
settings = Settings(openrouter_api_key="test-key", upload_dir=tmp_path)
|
|
|
|
with pytest.raises(SourceStorageError):
|
|
await store_source_file(filename="page.txt", file_bytes=b"text", settings=settings)
|