diff --git a/docs/Intent.md b/docs/Intent.md index c67dac7..7157db8 100644 --- a/docs/Intent.md +++ b/docs/Intent.md @@ -1,4 +1,4 @@ -# Historical Document Transcription +# Historical Document Transcription Design Intent I have several thousand pages of family history told through letters, postcards, books, and other documents that I want to transcribe to text. --- @@ -20,4 +20,5 @@ I have several thousand pages of family history told through letters, postcards, ## Methodology -See [transcription_methodology.md](transcription_methodology.md) for details on the transcription methodology. \ No newline at end of file +1. Follow current best practices per "A Guide to Documentary Editing" by Mary-Jo Kline. (See [Transcription Methodology](transcription_methodology.md)) + diff --git a/docs/V2 Python Pydantic Models.md b/docs/V2 Python Pydantic Models.md new file mode 100644 index 0000000..816f1d8 --- /dev/null +++ b/docs/V2 Python Pydantic Models.md @@ -0,0 +1,408 @@ +# SQLModel Table Models + +Each V2 table is represented by one `SQLModel` class. Because `SQLModel` is built on Pydantic and SQLAlchemy, these classes provide application validation and PostgreSQL mappings without parallel row and create models. + +Database-generated UUIDs and timestamps are `None` until PostgreSQL supplies their values during insert. The database columns remain non-nullable. `Person.metadata_` maps to the `metadata` column because `metadata` is reserved by SQLAlchemy's declarative API. + +```python +from datetime import date +from datetime import datetime +from enum import StrEnum +from uuid import UUID + +from pydantic import JsonValue +from sqlalchemy import Column +from sqlalchemy import Date +from sqlalchemy import DateTime +from sqlalchemy import ForeignKey +from sqlalchemy import Index +from sqlalchemy import Integer +from sqlalchemy import String +from sqlalchemy import Text +from sqlalchemy import UniqueConstraint +from sqlalchemy import text +from sqlalchemy.dialects.postgresql import JSONB +from sqlalchemy.dialects.postgresql import UUID as PostgreSQLUUID +from sqlmodel import Field +from sqlmodel import Relationship +from sqlmodel import SQLModel + + +class PersonRole(StrEnum): + AUTHOR = "author" + RECIPIENT = "recipient" + + +class JobStatus(StrEnum): + QUEUED = "queued" + PROCESSING = "processing" + COMPLETED = "completed" + PARTIAL_SUCCESS = "partial_success" + FAILED = "failed" + + +class JobSourceStatus(StrEnum): + PENDING = "pending" + TRANSCRIBED = "transcribed" + FAILED = "failed" + + +class Person(SQLModel, table=True): + __tablename__ = "person" + __table_args__ = (Index("idx_person_full_name", "full_name"),) + + id: UUID | None = Field( + default=None, + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + primary_key=True, + server_default=text("gen_random_uuid()"), + ), + ) + full_name: str = Field(sa_column=Column(Text, nullable=False)) + display_name: str | None = Field(default=None, sa_column=Column(Text)) + maiden_name: str | None = Field(default=None, sa_column=Column(Text)) + birth_date: date | None = Field(default=None, sa_column=Column(Date)) + birth_date_raw: str | None = Field(default=None, sa_column=Column(Text)) + birth_place: str | None = Field(default=None, sa_column=Column(Text)) + death_date: date | None = Field(default=None, sa_column=Column(Date)) + death_date_raw: str | None = Field(default=None, sa_column=Column(Text)) + death_place: str | None = Field(default=None, sa_column=Column(Text)) + biography: str | None = Field(default=None, sa_column=Column(Text)) + portrait_path: str | None = Field(default=None, sa_column=Column(Text)) + metadata_: JsonValue | None = Field( + default_factory=dict, + sa_column=Column( + "metadata", + JSONB, + server_default=text("'{}'::jsonb"), + ), + ) + created_at: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + updated_at: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + + document_people: list["DocumentPerson"] = Relationship( + back_populates="person", + sa_relationship_kwargs={"lazy": "raise", "passive_deletes": True}, + ) + + +class Document(SQLModel, table=True): + __tablename__ = "document" + __table_args__ = (Index("idx_document_date", "document_date"),) + + id: UUID | None = Field( + default=None, + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + primary_key=True, + server_default=text("gen_random_uuid()"), + ), + ) + name: str = Field(sa_column=Column(Text, nullable=False)) + document_type: str | None = Field(default=None, sa_column=Column(Text)) + document_date: date | None = Field(default=None, sa_column=Column(Date)) + document_date_raw: str | None = Field(default=None, sa_column=Column(Text)) + location_created: str | None = Field(default=None, sa_column=Column(Text)) + notes: str | None = Field(default=None, sa_column=Column(Text)) + archive_identifier: str | None = Field(default=None, sa_column=Column(Text)) + created_at: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + updated_at: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + + document_people: list["DocumentPerson"] = Relationship( + back_populates="document", + sa_relationship_kwargs={"lazy": "raise", "passive_deletes": True}, + ) + jobs: list["Job"] = Relationship( + back_populates="document", + sa_relationship_kwargs={"lazy": "raise", "passive_deletes": True}, + ) + sources: list["Source"] = Relationship( + back_populates="document", + sa_relationship_kwargs={"lazy": "raise", "passive_deletes": True}, + ) + + +class DocumentPerson(SQLModel, table=True): + __tablename__ = "document_person" + __table_args__ = ( + UniqueConstraint( + "document_id", + "person_id", + "role", + name="unique_document_person_role", + ), + Index("idx_document_person_doc", "document_id"), + Index("idx_document_person_per", "person_id"), + ) + + id: UUID | None = Field( + default=None, + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + primary_key=True, + server_default=text("gen_random_uuid()"), + ), + ) + document_id: UUID = Field( + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + ForeignKey("document.id", ondelete="CASCADE"), + nullable=False, + ), + ) + person_id: UUID = Field( + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + ForeignKey("person.id", ondelete="CASCADE"), + nullable=False, + ), + ) + role: PersonRole = Field(sa_column=Column(String(20), nullable=False)) + created_at: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + + document: Document | None = Relationship( + back_populates="document_people", + sa_relationship_kwargs={"lazy": "raise"}, + ) + person: Person | None = Relationship( + back_populates="document_people", + sa_relationship_kwargs={"lazy": "raise"}, + ) + + +class Job(SQLModel, table=True): + __tablename__ = "job" + __table_args__ = (Index("idx_job_document", "document_id"),) + + id: UUID | None = Field( + default=None, + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + primary_key=True, + server_default=text("gen_random_uuid()"), + ), + ) + document_id: UUID = Field( + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + ForeignKey("document.id", ondelete="CASCADE"), + nullable=False, + ), + ) + status: JobStatus = Field( + default=JobStatus.QUEUED, + sa_column=Column( + String(50), + nullable=False, + server_default=text("'queued'"), + ), + ) + retry_count: int = Field( + default=0, + sa_column=Column( + Integer, + nullable=False, + server_default=text("0"), + ), + ) + provider: str = Field(sa_column=Column(Text, nullable=False)) + model: str = Field(sa_column=Column(Text, nullable=False)) + prompt_name: str | None = Field(default=None, sa_column=Column(Text)) + date_created: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + date_updated: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + + document: Document | None = Relationship( + back_populates="jobs", + sa_relationship_kwargs={"lazy": "raise"}, + ) + job_sources: list["JobSource"] = Relationship( + back_populates="job", + sa_relationship_kwargs={"lazy": "raise", "passive_deletes": True}, + ) + + +class Source(SQLModel, table=True): + __tablename__ = "source" + __table_args__ = ( + Index("idx_source_document", "document_id"), + Index("idx_source_page_order", "document_id", "page_number"), + ) + + id: UUID | None = Field( + default=None, + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + primary_key=True, + server_default=text("gen_random_uuid()"), + ), + ) + document_id: UUID = Field( + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + ForeignKey("document.id", ondelete="CASCADE"), + nullable=False, + ), + ) + page_number: int = Field( + default=1, + sa_column=Column( + Integer, + nullable=False, + server_default=text("1"), + ), + ) + upload_name: str = Field(sa_column=Column(Text, nullable=False)) + filename: str = Field(sa_column=Column(Text, nullable=False)) + file_path: str = Field(sa_column=Column(Text, nullable=False)) + raw_transcription: str | None = Field(default=None, sa_column=Column(Text)) + revised_text: str | None = Field(default=None, sa_column=Column(Text)) + date_uploaded: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + date_revised: datetime | None = Field( + default=None, + sa_column=Column(DateTime(timezone=True)), + ) + + document: Document | None = Relationship( + back_populates="sources", + sa_relationship_kwargs={"lazy": "raise"}, + ) + job_sources: list["JobSource"] = Relationship( + back_populates="source", + sa_relationship_kwargs={"lazy": "raise", "passive_deletes": True}, + ) + + +class JobSource(SQLModel, table=True): + __tablename__ = "job_source" + __table_args__ = ( + UniqueConstraint("job_id", "source_id", name="unique_job_source"), + Index("idx_job_source_job", "job_id"), + Index("idx_job_source_source", "source_id"), + Index( + "idx_job_source_ai_metadata", + "ai_metadata", + postgresql_using="gin", + ), + ) + + id: UUID | None = Field( + default=None, + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + primary_key=True, + server_default=text("gen_random_uuid()"), + ), + ) + job_id: UUID = Field( + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + ForeignKey("job.id", ondelete="CASCADE"), + nullable=False, + ), + ) + source_id: UUID = Field( + sa_column=Column( + PostgreSQLUUID(as_uuid=True), + ForeignKey("source.id", ondelete="CASCADE"), + nullable=False, + ), + ) + status: JobSourceStatus = Field( + default=JobSourceStatus.PENDING, + sa_column=Column( + String(50), + nullable=False, + server_default=text("'pending'"), + ), + ) + raw_transcription: str | None = Field(default=None, sa_column=Column(Text)) + ai_metadata: JsonValue | None = Field( + default=None, + sa_column=Column(JSONB), + ) + raw_api_response: JsonValue | None = Field( + default=None, + sa_column=Column(JSONB), + ) + error_detail: str | None = Field(default=None, sa_column=Column(Text)) + executed_at: datetime | None = Field( + default=None, + sa_column=Column( + DateTime(timezone=True), + nullable=False, + server_default=text("now()"), + ), + ) + + job: Job | None = Relationship( + back_populates="job_sources", + sa_relationship_kwargs={"lazy": "raise"}, + ) + source: Source | None = Relationship( + back_populates="job_sources", + sa_relationship_kwargs={"lazy": "raise"}, + ) +``` + +The enum annotations validate application values while the mapped columns retain the `VARCHAR` types specified by the DDL. PostgreSQL owns generated UUIDs and timestamps through `server_default`; call `session.refresh(instance)` after a flush or commit when those generated values are needed immediately. + +`ai_metadata`, `raw_api_response`, and `metadata_` accept any JSON value supported by `JSONB`. Validate provider-specific payload structure before assigning it to these fields, while preserving the complete raw response in `raw_api_response`. + +Relationships use `lazy="raise"` to prevent implicit database I/O in async code. Queries must explicitly load relationships they need, for example with `selectinload()`. \ No newline at end of file diff --git a/docs/architecture.md b/docs/architecture.md deleted file mode 100644 index bd68d96..0000000 --- a/docs/architecture.md +++ /dev/null @@ -1,130 +0,0 @@ -# Architecture (V1 Baseline) - -This document describes the current architecture of the personal historical-document transcription system and serves as the V1 technical baseline. - -## Architecture Objectives - -- preserve source material as transcribed text -- keep operational complexity low for personal-scale deployment -- support asynchronous processing without external queue infrastructure -- maintain clear module boundaries for incremental extension - -## Runtime Topology - -V1 runtime is a modular monolith: - -- one FastAPI + NiceGUI application process -- one in-process async worker loop -- relational persistence via SQLModel (SQLite baseline) - -```mermaid -flowchart LR - U[Browser User] --> A[FastAPI + NiceGUI App] - A --> W[In-process Worker] - A --> DB[(SQLite via SQLModel)] - W --> P[OpenRouter Provider] - W --> DB -``` - -## Lifecycle Ownership - -Application lifespan owns runtime setup/teardown: - -- configure logging -- initialize and dispose DB runtime resources -- optional schema bootstrap by environment policy -- recover stale processing jobs -- start/stop worker consumer lifespan - -## Layered Module Structure - -### Interface Layer - -- `src/transcription/ui/**` (NiceGUI pages/components) -- `src/transcription/api/**` (FastAPI routes and error handlers) - -### Application/Workflow Layer - -- `src/transcription/services/workflows.py` -- `src/transcription/worker.py` - -Responsibilities: - -- orchestration and status transitions -- retry/timeout behavior -- provider call coordination - -### Service Layer - -- `src/transcription/services/*.py` - -Responsibilities: - -- CRUD and transactional boundaries -- domain-aligned persistence operations - -### Infrastructure Layer - -- `src/transcription/db/**` (runtime/session/bootstrap) -- `src/transcription/providers/**` (OpenRouter adapter) - -## Processing Workflow - -1. User uploads a source file from the UI. -2. App persists `Document`, `Job(queued)`, and `Source`. -3. Worker claims next queued job and marks `processing`. -4. Worker calls provider with prompt + source bytes. -5. On success, app writes immutable `Job.text` and marks `transcribed`. -6. On failure, app writes `Job.error_detail` and marks `failed`. -7. UI exposes job detail, original transcription, and optional revision. - -## Domain Ownership Invariants - -- `Job.text` is immutable original provider output. -- `Revision` is optional, user-authored, and linked to `Source`. -- `Revision` does not overwrite original job transcription. -- Status lifecycle is fixed to: `queued -> processing -> transcribed|failed`. - -## Data Model Summary - -- `Document` has many `Source` and many `Job`. -- `Source` belongs to one `Document` and one `Job`. -- `Source` has optional `Revision` (`0..1`) enforced by unique `revision.source_id`. - -## Simplicity Guardrails (V1) - -- no external queue/broker required -- no search engine required -- no distributed worker fleet required -- keep provider integration behind adapter boundary - -## Extension Path - -### V1 (current) - -- SQLite baseline -- OpenRouter provider -- in-process worker -- optional single revision workflow - -### V2 (planned) - -- PostgreSQL as relational baseline -- optional MongoDB adjunct store for scoped use cases -- migration-first schema evolution - -See [ver2/ver2.md](ver2/ver2.md) for roadmap details. - -## Test Strategy - -- unit tests for model/service behaviors -- integration tests for upload/workflow reliability -- UI integration tests for page/render contracts -- external provider tests opt-in via marker/config - -## Related References - -- [index.md](index.md) -- [requirements.md](requirements.md) -- [schema.md](schema.md) -- [error_handling.md](error_handling.md) diff --git a/docs/architecture_v2.md b/docs/architecture_v2.md new file mode 100644 index 0000000..1f55a8c --- /dev/null +++ b/docs/architecture_v2.md @@ -0,0 +1,136 @@ +# System Architecture (Version 2) + +This document describes the V2 production architecture of the personal historical-document transcription system. + +## Architecture Objectives + +* Preserve source material as immutable transcribed text alongside page-level spatial AI metadata. +* Support batching multi-image and folder uploads cleanly into sequential pages (`page_number`). +* Leverage asynchronous worker pools (`asyncio`) for parallel single-image API execution bounded by rate limiters (`asyncio.Semaphore`). +* Migrate persistence to PostgreSQL using native `UUID`, `TIMESTAMPTZ`, and `JSONB` document storage. +* Standardize all data validation, API parsing, and database models on **Pydantic V2**. +* Support rich historical attribution (multi-author and multi-recipient relationships). + +## Runtime Topology + +The V2 runtime operates as an asynchronous Python application: + +* FastAPI + NiceGUI web application process. +* In-process `asyncio` background task orchestrator for parallel API execution. +* Relational persistence via PostgreSQL (using `asyncpg` or `psycopg3`). +* Pydantic V2 validation layer wrapping API payloads and PostgreSQL `JSONB` schemas. + +```mermaid +flowchart LR +U[Browser User] --> A[FastAPI + NiceGUI App] +A --> W[Asyncio Worker Engine] +A --> DB[(PostgreSQL Database)] +W --> P[Vision Provider APIs\nOpenAI / Claude] +W --> DB +``` + +## Lifecycle Ownership + +Application lifespan owns runtime setup/teardown: + +* Initialize environment logging and Pydantic configuration. +* Manage asynchronous PostgreSQL connection pools (`asyncpg` / `psycopg3`). +* Execute database migrations and index initialization. +* Recover stale processing jobs on startup. +* Manage graceful shutdown of active `asyncio` worker pools. + +## Layered Module Structure + +### Interface Layer + +* `src/transcription/ui/**` (NiceGUI pages, multi-page renderers, person cards) +* `src/transcription/api/**` (FastAPI routes and JSON error handlers) + +### Application & Async Worker Layer + +* `src/transcription/services/workflows.py` +* `src/transcription/worker.py` + +Responsibilities: + +* Batch orchestration and status transitions (`queued` -> `processing` -> `completed` | `partial_success` | `failed`). +* Parallel single-image API execution using `asyncio.gather` bounded by `asyncio.Semaphore`. +* Pydantic schema parsing (`PageAIMetadata`) and validation prior to database storage. + +### Domain & Service Layer + +* `src/transcription/models/*.py` (Pydantic V2 schemas and entity definitions) +* `src/transcription/services/*.py` (Transactional operations for `Document`, `Person`, `Source`, `Job`, and `JobSource`) + +### Infrastructure Layer + +* `src/transcription/db/**` (PostgreSQL connection pooling and raw parameterized SQL execution) +* `src/transcription/providers/**` (OpenAI & Anthropic Vision SDK adapters) + +## Processing Workflow + +1. User uploads a folder or batch of images for a `Document`. +2. System creates `Document`, `Job(status='queued')`, and ordered `Source` pages (`page_number = 1..N`). +3. Worker claims job, sets `Job.status = 'processing'`, and spawns parallel `asyncio` tasks bounded by semaphore. +4. Each task calls Vision API for a **single** `Source` image. +5. On task completion: +* Writes a `JobSource` record containing `status='transcribed'`, `raw_transcription`, `ai_metadata` (bounding boxes/confidence), and `raw_api_response`. +* Caches active text to `Source.raw_transcription`. + + +6. On page failure: +* Writes `JobSource` record with `status='failed'` and `error_detail`. + + +7. Once all page tasks resolve: +* Marks `Job.status` as `completed` (100% success), `partial_success` (at least 1 success, 1 failure), or `failed` (all failed). + + + +## Domain Ownership & Invariants + +* **Immutable AI Outputs:** `source.raw_transcription` and `job_source.raw_transcription` store original, point-in-time machine output and are immutable. +* **Inlined Revisions:** Human corrections occur on `source.revised_text`. UI renders `COALESCE(revised_text, raw_transcription)`. +* **Sequential Integrity:** Multi-page documents are strictly ordered by `source.page_number ASC`. +* **Page Execution Isolation:** A failure on one page image does not invalidate successful transcriptions on sister pages in the same batch job. + +## Data Model Summary + +* `Document` has many `Source` pages, many `Job` runs, and many `Person` records via `DocumentPerson` junction (`author` or `recipient`). +* `Source` belongs to one `Document` and can be processed across many `JobSource` executions. +* `Job` has many `JobSource` execution records. + +## Test Strategy + +* Unit tests for Pydantic V2 schemas, custom validators, and JSONB serialization. +* Integration tests for async PostgreSQL connection handling and parameterized queries. +* Async workflow tests using mock AI providers to verify `partial_success` and retry logic. +* UI integration tests for multi-page rendering and person management. + +--- + +## Technology References + +- [FastAPI documentation](https://fastapi.tiangolo.com/) +- [NiceGUI documentation](https://nicegui.io/documentation) +- [PostgreSQL documentation](https://www.postgresql.org/docs/) +- [Python asyncio](https://docs.python.org/3/library/asyncio.html#module-asyncio) +- [Pydantic Validation](https://pydantic.dev/docs/validation/latest/get-started/) +- [Pydantic AI](https://pydantic.dev/docs/ai/overview/) + + +## Related Local References + +- [System Overview](index_v1.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- System Architecture (this document) +- [System Requirements](requirements_v1.md) +- [Data model](schema_v1.md) +- [Error Handling Policy](error_handling_v1.md) +- [Implementation Plan](implementation_plan_v1.md) + + + + + diff --git a/docs/archive/README.md b/docs/archive/README.md deleted file mode 100644 index c29d558..0000000 --- a/docs/archive/README.md +++ /dev/null @@ -1,23 +0,0 @@ -# V2 Archive - -This folder preserves pre-V1-alignment versions of core documentation that included planned target-state architecture material. - -Archived snapshots: - -- `index.pre-v1-alignment.md` -- `requirements.pre-v1-alignment.md` -- `architecture.pre-v1-alignment.md` - -Purpose: - -- keep a durable reference for planned architecture language -- reduce risk of losing useful V2 direction while V1 docs stay implementation-aligned - -Notes: - -- These files are historical snapshots, not the active V1 source of truth. -- Active V1 docs remain at: - - `docs/index.md` - - `docs/requirements.md` - - `docs/architecture.md` -- V2 planning should continue in `docs/ver2/ver2.md` and related V2 artifacts. diff --git a/docs/ver2/V2 PostgreSQL DDL Specification.md b/docs/ddl_v2.sql similarity index 98% rename from docs/ver2/V2 PostgreSQL DDL Specification.md rename to docs/ddl_v2.sql index 9eaf912..65c1d75 100644 --- a/docs/ver2/V2 PostgreSQL DDL Specification.md +++ b/docs/ddl_v2.sql @@ -1,4 +1,4 @@ -## PostgreSQL DDL Specification +## PostgreSQL DDL Specification (Version 2) ```sql -- Enable pgcrypto for UUID generation if on PostgreSQL < 13 diff --git a/docs/error_handling_v2.md b/docs/error_handling_v2.md new file mode 100644 index 0000000..5325c56 --- /dev/null +++ b/docs/error_handling_v2.md @@ -0,0 +1,88 @@ +# Error Handling Policy (Version 2) + +This document defines the canonical error-handling policy for the V2 document transcription system. + +## Error Handling Objectives + +* Make failures visible in clear, actionable language at both the document and individual page levels. +* Support **isolated failure handling** in multi-image batches so single page errors do not crash an entire batch job. +* Preserve diagnostic detail (Pydantic validation errors, raw provider responses) in PostgreSQL `JSONB` for fast troubleshooting. +* Ensure consistent error envelope structure across API, UI, and async worker boundaries. + +## Scope And Authority + +Governs error behavior across NiceGUI pages, FastAPI routes, service orchestration, `asyncio` background tasks, PostgreSQL interactions, and AI provider adapters. + +## Error Taxonomy + +| Category | Definition | Retriable | +| --- | --- | --- | +| `validation_error` | Pydantic payload or parameter schema validation failure | no | +| `user_input_error` | Unacceptable user file (unsupported image type, corrupt file) | no | +| `not_found_error` | Requested resource (`Document`, `Source`, `Person`, `Job`) missing | no | +| `conflict_error` | Operation violates state constraints (e.g., duplicate `document_person` role) | no | +| `external_provider_error` | AI Provider API failure (rate limit, vision execution error) | yes | +| `infrastructure_transient_error` | Temporary DB connection reset or HTTP timeout | yes | +| `infrastructure_persistent_error` | Database down, missing API credentials, misconfiguration | no | +| `internal_unexpected_error` | Uncaught Python exception or logic defect | no | + +## Async Batch & Page-Level Error Behavior + +In multi-image `asyncio` batch processing: + +1. **Page Isolation:** Exceptions caught during individual page calls are caught within the `asyncio` task wrapper. +2. **Page Record Logging:** Page failure detail is written directly to `job_source.error_detail` and `job_source.status = 'failed'`. +3. **Batch Aggregate State:** +* If **all** page tasks succeed -> `job.status = 'completed'`. +* If **some** page tasks fail -> `job.status = 'partial_success'`. +* If **all** page tasks fail -> `job.status = 'failed'`. + + +4. **Retry Strategy:** The UI exposes a "Retry Failed Pages" option for `partial_success` jobs, which spawns a new targeted `Job` containing *only* the `Source` IDs marked as `failed`. + +## API Error Response Contract + +API error responses return a structured JSON envelope: +```json +{ +"error_id": "err_uuid_12345", +"category": "validation_error", +"message": "The uploaded payload failed schema validation.", +"suggestion": "Check file format and metadata fields, then try again.", +"details": { +"pydantic_errors": [...] +}, +"timestamp": "2026-07-31T07:55:00Z" +} +``` + +HTTP Status Mappings: + +* `validation_error`, `user_input_error` -> `400` +* `not_found_error` -> `404` +* `conflict_error` -> `409` +* `external_provider_error` -> `502` / `503` +* `infrastructure_transient_error` -> `503` +* `infrastructure_persistent_error`, `internal_unexpected_error` -> `500` + +--- + +## Technology References + +- [FastAPI documentation](https://fastapi.tiangolo.com/) +- [NiceGUI documentation](https://nicegui.io/documentation) +- [PostgreSQL documentation](https://www.postgresql.org/docs/) +- [Python asyncio](https://docs.python.org/3/library/asyncio.html#module-asyncio) +- [Pydantic Validation](https://pydantic.dev/docs/validation/latest/get-started/) +- [Pydantic AI](https://pydantic.dev/docs/ai/overview/) + +## Related Local References + +- [System Overview](index_v2.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v2.md) +- [System Requirements](requirements_v2.md) +- [Data model](schema_v2.md) +- Error Handling Policy (this document) +- [Implementation Plan](implementation_plan_v2.md) diff --git a/docs/implementation_plan_v2.md b/docs/implementation_plan_v2.md new file mode 100644 index 0000000..17b95b0 --- /dev/null +++ b/docs/implementation_plan_v2.md @@ -0,0 +1,248 @@ +# Implementation Plan (Version 2) + +This plan defines the path from the V1 baseline to **Version 2 complete**, aligned to the updated multi-image and multi-person relational domain model: + +* `Document` acts as a logical parent container for physical artifacts, supporting multi-author and multi-recipient relationships via `DocumentPerson`. +* `Source` represents an individual image page within a document, maintaining sequential order (`page_number`), cached active machine output (`raw_transcription`), and inline single user revisions (`revised_text`). +* `Job` acts as an overarching batch orchestrator for multi-page async processing tasks. +* `JobSource` records individual point-in-time API executions per image page, storing Pydantic-validated `ai_metadata` and raw REST envelopes (`raw_api_response`). +* **Pydantic V2** acts as the single source of truth for runtime validation, API payload parsing, and PostgreSQL JSONB serialization. + +The objective is to complete the V2 scope with production readiness while keeping non-V2 enhancements out of active delivery. + +--- + +## V2 Completion Definition + +V2 is complete when all of the following are true: + +1. **Functional complete** +* Multi-image and whole-folder uploads assign sequential page numbers to `Source` records under a single `Document`. +* Batch jobs process pages concurrently using an `asyncio` worker pool with semaphore rate limiting. +* Partial job failures resolve cleanly to `partial_success`, allowing single-page retries without re-running successful pages. +* Multi-author and multi-recipient tagging is supported on `Document`. + + +2. **Data-model complete** +* SQLite is fully replaced with PostgreSQL (using `asyncpg` or `psycopg3`). +* Pydantic V2 models validate all API payloads, database row mappings, and `JSONB` structures. + + +3. **Operational complete** +* Concurrency controls, worker pool metrics, and database connections operate safely under batch load. + + +4. **Documentation complete** +* `schema_v2.md`, `DDL_v2.sql`, Pydantic model contracts are updated and consistent. + + + +--- + +## Phase 1 — Data Contract Stabilization & Pydantic Baseline + +**Goal:** Lock the PostgreSQL schema, DDL, and Pydantic V2 models before refactoring service logic. + +### Tasks + +1. Finalize DDL for PostgreSQL native types (`UUID`, `TIMESTAMPTZ`, `JSONB`) and junction tables (`document_person`, `job_source`). +2. Build core Pydantic V2 schemas (`Person`, `Document`, `Source`, `Job`, `JobSource`, `PageAIMetadata`). +3. Confirm and document data invariants: +* `source.raw_transcription` and `job_source.raw_transcription` are immutable machine outputs. +* `source.revised_text` holds user edits. UI renders `COALESCE(revised_text, raw_transcription)`. +* Page sequence is strictly ordered by `source.page_number ASC`. + + +4. Freeze V2 job status values (`queued`, `processing`, `completed`, `partial_success`, `failed`) and page execution status values (`pending`, `transcribed`, `failed`). + +### Deliverables + +* Canonical `docs/schema_v2.md` and `docs/DDL_v2.sql`. +* Centralized Pydantic validation suite in `models/schemas_v2.py`. + +### Exit Criteria + +* All database tables, relationships, and JSONB structures have corresponding Pydantic V2 models passing unit validation tests. + +--- + +## Phase 2 — Persistence Layer Transition (SQLite to PostgreSQL) + +**Goal:** Replace the SQLite storage layer with an asynchronous PostgreSQL driver (`asyncpg` or `psycopg3`). + +### Tasks + +1. Configure PostgreSQL database connection pooling and environment configuration. +2. Refactor `services/store.py` / repository layers to execute parameterized async SQL queries (`$1`, `$2`). +3. Implement JSONB serialization and deserialization helpers using Pydantic's `.model_dump_json()` and `.model_validate()`. +4. Implement database bootstrap routines for PostgreSQL table creation and index initialization. + +### Deliverables + +* PostgreSQL-native database connection and query service modules. +* Integration test suite confirming connection pooling and JSONB CRUD operations. + +### Exit Criteria + +* All database reads/writes run asynchronously against PostgreSQL with zero remaining SQLite driver dependencies. + +--- + +## Phase 3 — Service Layer & `asyncio` Engine Refactor + +**Goal:** Implement batch orchestration and parallel single-image API execution. + +### Tasks + +1. Refactor upload service to process folder/multi-image input: +* Group files into a single `Document`. +* Create ordered `Source` rows (`page_number = 1..N`). + + +2. Refactor `services/workflows.py` with `asyncio` worker pools: +* Use `asyncio.Semaphore` to enforce API provider rate limits. +* Issue parallel single-image requests to Vision APIs (OpenAI/Claude). +* Parse API responses directly into Pydantic models (`PageAIMetadata`). + + +3. Update execution tracking: +* Create a `JobSource` row per page call to record `raw_transcription`, `ai_metadata`, and `raw_api_response`. +* Update active `source.raw_transcription` upon task completion. +* Calculate aggregate batch status (`completed`, `partial_success`, `failed`) on the parent `Job`. + + +4. Refactor `services/person.py` and `services/documents.py` to handle multi-person roles via `document_person`. + +### Deliverables + +* Asynchronous batch execution engine in `services/workflows.py`. +* Service routines for multi-person tagging and page-level retries. + +### Exit Criteria + +* Executing a folder upload of 10+ images processes concurrently, populates page-level `JobSource` entries, and handles partial worker errors without crashing the batch. + +--- + +## Phase 4 — UI & API Contract Alignment + +**Goal:** Update API endpoints and frontend/UI views to render multi-page documents and person roles. + +### Tasks + +1. Update document and job API endpoints to accept batch file arrays and multi-person ID payloads. +2. Update UI document views: +* Render multi-page document transcriptions sequentially by `page_number`. +* Display author and recipient chips/cards linked from `document_person`. + + +3. Update job detail UI to show page-level execution statuses (`transcribed` vs. `failed`) and provide a "Retry Failed Pages" action for `partial_success` jobs. +4. Align inline page editing controls to update `source.revised_text` and `source.date_revised`. + +### Deliverables + +* Refactored API routes and UI components supporting multi-page rendering and person management. + +### Exit Criteria + +* UI successfully displays multi-page document text, allows per-page human revisions, and shows author/recipient metadata. + +--- + +## Phase 5 — Test Suite Realignment & Concurrency Testing + +**Goal:** Ensure end-to-end system stability under concurrent async execution and load. + +### Tasks + +1. Write unit tests for Pydantic models, custom validators, and JSONB conversions. +2. Write integration tests for async database operations: +* CRUD for `Document`, `Person`, `DocumentPerson`, `Source`, `Job`, and `JobSource`. + + +3. Write mock-backed async workflow tests: +* Verify `asyncio.Semaphore` bounds concurrent tasks properly. +* Validate state transition logic for `completed`, `partial_success`, and `failed` jobs. +* Confirm retry routines process only targeted `JobSource` records marked as `failed`. + + +4. Re-enable CI quality gates (linting, type checking with Pyright/mypy, pytest). + +### Deliverables + +* Passing asynchronous test suite covering core workflows, edge cases, and failure recoveries. + +### Exit Criteria + +* CI pipeline is green with comprehensive coverage across database operations, Pydantic models, and worker queues. + +--- + +## Phase 6 — Reliability, Operations, and Release Readiness + +**Goal:** Prepare V2 for production deployment and operator management. + +### Tasks + +1. Verify structured logging includes `job_id`, `document_id`, `source_id`, and `person_id`. +2. Tune PostgreSQL connection pool limits and `asyncio` concurrency thresholds for production infrastructure. +3. Update operational documentation: +* Review and update `docs/schema_v2.md` as needed. +* Create `docs/runbook_v2.md` detailing PostgreSQL maintenance, JSONB index management, and worker queue monitoring. +* Create `docs/release_checklist_v2.md` for launch sign-off. + + + +### Deliverables + +* Updated project documentation and operational runbooks. +* V2 release sign-off checklist. + +### Exit Criteria + +* All documentation reflects V2 architecture; launch checklist is fully verified. + +--- + +## Requirement Traceability Focus + +Maintain evidence against these V2 requirement groups: + +* **Batch & Multi-Image Pipeline:** Folder ingestion, page ordering, async worker execution. +* **Database & Persistence:** PostgreSQL, native UUIDs, JSONB execution storage, `asyncpg` pooling. +* **Validation & Schemas:** Pydantic V2 models for DB rows, API requests, and AI vision responses. +* **Attribution & Metadata:** Multi-author and multi-recipient tagging, biographical entity management. +* **Error Recovery:** Partial success states, page-level status flags, isolated retry execution. + +--- + +## Scope Discipline Rule (V2 Focus) + +* Only tasks required for V2 scope (PostgreSQL, Pydantic V2, folder/async processing, multi-person roles) enter this plan. +* V3 candidate features (such as side-by-side multi-provider model output comparison) remain strictly in the future backlog. +* Any schema adjustments during implementation require immediate updates to `DDL_v2.sql`, Pydantic models, and `schema_v2.md`. + +--- + +## Technology References + +- [FastAPI documentation](https://fastapi.tiangolo.com/) +- [NiceGUI documentation](https://nicegui.io/documentation) +- [PostgreSQL documentation](https://www.postgresql.org/docs/) +- [Python asyncio](https://docs.python.org/3/library/asyncio.html#module-asyncio) +- [Pydantic Validation](https://pydantic.dev/docs/validation/latest/get-started/) +- [Pydantic AI](https://pydantic.dev/docs/ai/overview/) + +## Related Local References + +- [System Overview](index_v2.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v2.md) +- [System Requirements](requirements_v2.md) +- [Data model](schema_v2.md) +- [Error Handling Policy](error_handling_v2.md) +- Implementation Plan (this document) + + + diff --git a/docs/index.md b/docs/index.md deleted file mode 100644 index 380e555..0000000 --- a/docs/index.md +++ /dev/null @@ -1,57 +0,0 @@ -## Document Transcription System (V1) - -This project is a personal-scale application for transcribing and preserving historical family documents. - -## Start Here - -Read [architecture.md](architecture.md) first. - -The architecture page is the primary technical reference for: - -- runtime topology and infrastructure assumptions -- module boundaries and dependency flow -- processing lifecycle and data ownership -- test strategy and extension path - -## What The Application Does - -At a high level, users upload images/PDFs, jobs are processed asynchronously, and users review original transcriptions plus optional revisions. - -Core V1 capabilities: - -- upload supported source files (`.jpg`, `.jpeg`, `.png`, `.tif`, `.tiff`, `.pdf`) -- asynchronous job processing with visible status (`queued`, `processing`, `transcribed`, `failed`) -- immutable original transcription stored on `Job.text` -- optional single user-authored revision per source (`0..1`) -- prompt artifacts stored as Markdown files in `prompts/` - -## Current Operating Model (V1 Baseline) - -- application service: FastAPI + NiceGUI -- persistence baseline: SQLModel with SQLite -- worker: in-process async background loop -- deployment baseline: lightweight Docker Compose app runtime - -> Planned persistence evolution (PostgreSQL and optional MongoDB) belongs to V2 planning and is tracked separately. - -## Documentation Map - -- Architecture and technical design: [architecture.md](architecture.md) -- V1 runtime and requirement baseline: [requirements.md](requirements.md) -- Data model and constraints: [schema.md](schema.md) -- Error handling policy and operational guidance: [error_handling.md](error_handling.md) -- V1 requirement evidence matrix: [traceability_v1.md](traceability_v1.md) -- Operations runbook: [runbook.md](runbook.md) -- V1 migration and rollback guidance: [migration_v1.md](migration_v1.md) -- V1 release checklist: [release_checklist_v1.md](release_checklist_v1.md) -- Domain context and transcription policy: [intent.md](intent.md) -- Transcription methodology: [transcription_methodology.md](transcription_methodology.md) -- V1 execution plan: [ver1/ver1.md](ver1/ver1.md) -- V2 roadmap: [ver2/ver2.md](ver2/ver2.md) - -## Glossary - -- Prompt artifact: a Markdown file containing one transcription prompt. -- Original transcription: immutable provider output stored on `Job.text`. -- Revision: optional user-authored text linked to a `Source`. -- System of record: the authoritative persistent store for canonical application data. diff --git a/docs/index_v2.md b/docs/index_v2.md new file mode 100644 index 0000000..de92270 --- /dev/null +++ b/docs/index_v2.md @@ -0,0 +1,47 @@ +# Document Transcription System Overview (Version 2) + +This project is a personal-scale application for transcribing, indexing, and preserving historical family documents, letters, postcards, and journals. + +## Start Here + +Read [architecture_v2.md](architecture_v2.md) first for technical overview and system design. + +## Core V2 Capabilities + +* **Folder & Multi-Image Ingestion:** Upload whole folders or image batches that map sequentially (`page_number`) under a single `Document`. +* **Parallel Async AI Vision Engine:** Concurrently process single-page image transcriptions using Python `asyncio` bounded by rate limiters. +* **Robust PostgreSQL Storage:** Relational storage for entities with native `UUID`, `TIMESTAMPTZ`, and `JSONB` for deep AI spatial metadata and raw envelopes. +* **Pydantic V2 Validation:** End-to-end type safety, DB row mapping, and JSONB payload validation. +* **Historical Person Management:** Track authors and recipients across documents with rich biographical entities (`Person`). +* **Page-Level Execution Auditing & Revisions:** Store immutable point-in-time machine output per run while enabling inline human corrections (`revised_text`). +* **Partial Failure Recovery:** Bounded batch execution that isolates single-page API errors (`partial_success`) for simple retries. + +## Technical Stack + +* **Application Web Framework:** FastAPI + NiceGUI +* **Persistence Engine:** PostgreSQL 13+ +* **Data Validation & Schemas:** Pydantic V2 +* **Concurrency & Workers:** Python `asyncio` worker pool with `asyncio.Semaphore` +* **Vision Providers:** OpenAI (GPT-4o) and Anthropic (Claude 3.5 Sonnet) via native SDKs + +--- + +## Technology References + +- [FastAPI documentation](https://fastapi.tiangolo.com/) +- [NiceGUI documentation](https://nicegui.io/documentation) +- [PostgreSQL documentation](https://www.postgresql.org/docs/) +- [Python asyncio](https://docs.python.org/3/library/asyncio.html#module-asyncio) +- [Pydantic Validation](https://pydantic.dev/docs/validation/latest/get-started/) +- [Pydantic AI](https://pydantic.dev/docs/ai/overview/) + +## Documentation Index + +- System Overview (this document) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v2.md) +- [System Requirements](requirements_v2.md) +- [Data model](schema_v2.md) +- [Error Handling Policy](error_handling_v2.md) +- [Implementation Plan](implementation_plan_v2.md) diff --git a/docs/requirements.md b/docs/requirements.md deleted file mode 100644 index edf8436..0000000 --- a/docs/requirements.md +++ /dev/null @@ -1,86 +0,0 @@ -## Document Transcription System Requirements (V1 Baseline) - -This page captures the **Version 1 baseline requirements** for the currently implemented system. It is the source of truth for V1 acceptance and test traceability. - -Forward-looking architecture changes (for example PostgreSQL/Mongo adoption) are intentionally out of this document and should be tracked in a V2 planning/backlog artifact. - -## Scope - -- System of interest: a single Python application service (NiceGUI + FastAPI) with SQLModel persistence. -- Runtime/persistence baseline: local-first execution using SQLite (default `sqlite:///./transcription.db`), with Docker Compose support. -- Primary concern: end-to-end transcription lifecycle from upload through terminal state plus optional single revision editing. - -## Requirements Model (Concise Text Form) - -### Requirements - -| ID | Category | Requirement | Risk | Verify Method | -| --- | --- | --- | --- | --- | -| REQ-0 | System | Provide end-to-end document transcription with persistent, inspectable lifecycle state. | medium | demonstration | -| REQ-1 | Functional | Allow users to upload supported image/PDF files as sources from the web UI. | low | test | -| REQ-2 | Functional | Process uploads asynchronously and return either original transcription output or explicit failure. | high | test | -| REQ-3 | Functional | Persist and expose job states: `queued`, `processing`, `transcribed`, `failed`. | high | inspection | -| REQ-4 | Functional | Persist original provider output (`Job.text`) and failure detail (`Job.error_detail`) for each job. | medium | test | -| REQ-5 | Interface | Expose API/UI views for status inspection and transcription reading. | medium | demonstration | -| REQ-6 | Performance | Trigger background processing on upload to preserve UI responsiveness. | medium | analysis | -| REQ-7 | Design Constraint | Keep lifespan-owned runtime resources (engine/session factory/worker resources) initialized and disposed at application boundaries. | medium | inspection | -| REQ-8 | Design Constraint | Initialize configuration and logging once at startup through centralized mechanisms. | low | inspection | -| REQ-9 | Design Constraint | Support containerized app runtime via Docker Compose using the same V1 persistence model. | medium | demonstration | -| REQ-10 | Design Constraint | Keep schema bootstrap explicit and opt-in for production safety. | high | inspection | -| REQ-11 | Design Constraint | Route persistence changes through service/workflow orchestration boundaries. | medium | inspection | -| REQ-12 | Design Constraint | Store transcription prompts as individual Markdown artifacts for iterative refinement. | medium | inspection | -| REQ-13 | Functional | Allow users to create/update one optional revision derived from the original job transcription and view/delete it from the job detail flow. | low | test | - -### Requirement Relationships - -- Contains: REQ-0 contains REQ-1 through REQ-13. -- Derives: REQ-2 -> REQ-3, REQ-3 -> REQ-4. -- Traces: REQ-5 -> REQ-3. -- Refines: REQ-6 -> REQ-2. - -### Architecture Elements - -| Element | Type | Doc Reference | -| --- | --- | --- | -| UI | NiceGUI pages/components | `src/transcription/ui/pages`, `src/transcription/ui/components` | -| API | FastAPI routes and handlers | `src/transcription/api`, `src/transcription/app.py` | -| WORKER | Async queued-job processing workflow | `src/transcription/worker.py`, `src/transcription/services/workflows.py` | -| DBREL | SQLModel relational persistence (SQLite in V1 baseline) | `src/transcription/models.py`, `src/transcription/db` | -| SERVICES | Service-layer persistence orchestration | `src/transcription/services` | -| OPS | Containerized runtime baseline | `docker-compose.yml`, `Dockerfile` | -| PROMPTS | Transcription prompt artifacts | `prompts/` | -| TESTS | Pytest verification suite | `tests/` | - -### Satisfaction Mapping - -- UI satisfies REQ-1, REQ-5, REQ-13. -- API satisfies REQ-5. -- WORKER satisfies REQ-2, REQ-6. -- DBREL satisfies REQ-3, REQ-4, REQ-10, REQ-13. -- SERVICES satisfies REQ-4, REQ-11. -- OPS satisfies REQ-9. -- PROMPTS satisfies REQ-12. - -### Verification Mapping - -- TESTS verifies REQ-1, REQ-2, REQ-3, REQ-4, REQ-5, REQ-10, REQ-11, REQ-12, REQ-13. - -## Requirement Notes - -- Requirement IDs (`REQ-*`) are stable references for planning, implementation, and traceability. -- This document is intentionally **implementation-aligned** for V1 completion and release sign-off. -- Planned storage evolution (PostgreSQL and optional MongoDB) is a **V2 concern** and should be tracked outside this V1 baseline. - -## Verification Intent - -- Demonstration: validate end-to-end behavior through operator-visible flows. -- Inspection: verify architecture and startup/runtime policies in code and configuration. -- Analysis: evaluate asynchronous execution behavior and design sufficiency. -- Test: automate behavioral checks through pytest suites and service/UI integration tests. - -## Glossary - -- Original transcription: immutable provider output stored on `Job.text`. -- Revision: optional user-authored editable text tied to a `Source` (`0..1` in V1). -- Prompt artifact: a Markdown file containing instructions used for transcription. -- System of record: the authoritative relational store for canonical V1 data. diff --git a/docs/requirements_v2.md b/docs/requirements_v2.md new file mode 100644 index 0000000..f21b3c0 --- /dev/null +++ b/docs/requirements_v2.md @@ -0,0 +1,42 @@ +# Document Transcription System Requirements (Version 2) + +This document captures the **Version 2 baseline requirements** for the production implementation. + +## Requirements Model + +| ID | Category | Requirement | Verify Method | +| --- | --- | --- | --- | +| REQ-0 | System | Provide end-to-end multi-page document transcription with persistent, inspectable async job states. | demonstration | +| REQ-1 | Functional | Allow users to upload folders or multi-image batches as sequential `Source` pages under a `Document`. | test | +| REQ-2 | Functional | Process multi-page jobs asynchronously using an `asyncio` worker pool bounded by rate limits. | test | +| REQ-3 | Functional | Persist page-level execution outputs (`raw_transcription`, `ai_metadata`, `raw_api_response`) on `JobSource`. | test | +| REQ-4 | Functional | Support job states (`queued`, `processing`, `completed`, `partial_success`, `failed`) and page states (`pending`, `transcribed`, `failed`). | inspection | +| REQ-5 | Functional | Allow users to manage historical `Person` records and link multiple authors/recipients to a `Document` via `DocumentPerson`. | test | +| REQ-6 | Functional | Maintain immutable original machine output on `Source.raw_transcription` while permitting inline human edits on `Source.revised_text`. | test | +| REQ-7 | Data Constraint | Store all persistent domain data in PostgreSQL using native `UUID`, `TIMESTAMPTZ`, and `JSONB` columns. | inspection | +| REQ-8 | Data Constraint | Validate all API requests, database rows, and JSONB structures using Pydantic V2 schemas. | test | +| REQ-9 | Interface | Render multi-page transcriptions sequentially by `page_number` in the web UI with author/recipient metadata. | demonstration | +| REQ-10 | Operations | Allow operators to retry only failed pages for jobs in a `partial_success` state. | test | + +## Element Satisfaction Mapping + +* **UI (NiceGUI):** Satisfies REQ-1, REQ-5, REQ-6, REQ-9, REQ-10. +* **API (FastAPI):** Satisfies REQ-1, REQ-4, REQ-5, REQ-8. +* **WORKER (asyncio):** Satisfies REQ-2, REQ-3, REQ-4, REQ-10. +* **PERSISTENCE (PostgreSQL):** Satisfies REQ-3, REQ-6, REQ-7. +* **MODELS (Pydantic V2):** Satisfies REQ-8. + +--- + +## Related Local References + +- [System Overview](index_v2.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v2.md) +- System Requirements (this document) +- [Data model](schema_v2.md) +- [Error Handling Policy](error_handling_v2.md) +- [Implementation Plan](implementation_plan_v2.md) + + diff --git a/docs/ver2/V2 DB Schema.md b/docs/schema_v2.md similarity index 90% rename from docs/ver2/V2 DB Schema.md rename to docs/schema_v2.md index e812c01..17408f2 100644 --- a/docs/ver2/V2 DB Schema.md +++ b/docs/schema_v2.md @@ -1,4 +1,4 @@ -# Database Schema (V2 Architecture) +# Database Schema (Version 2) This document describes the PostgreSQL relational schema for the transcription platform. It incorporates multi-image batch orchestration via `asyncio`, page-level execution tracking, many-to-many author/recipient attribution, and JSONB document storage for AI vision outputs. @@ -119,4 +119,18 @@ erDiagram ### Attribution & Person Roles * Multi-Person Roles: Documents support zero, one, or many authors and recipients linked via document_person. -* Role Uniqueness: (document_id, person_id, role) must be unique to prevent duplicate role tagging. \ No newline at end of file +* Role Uniqueness: (document_id, person_id, role) must be unique to prevent duplicate role tagging. + +--- + +## Related Local References + +- [System Overview](index_v2.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v2.md) +- [System Requirements](requirements_v2.md) +- Data model (this document) +- [Error Handling Policy](error_handling_v2.md) +- [Implementation Plan](implementation_plan_v2.md) + diff --git a/docs/archive/architecture.pre-v1-alignment.md b/docs/ver1/architecture_v1.md similarity index 96% rename from docs/archive/architecture.pre-v1-alignment.md rename to docs/ver1/architecture_v1.md index dd632ec..ace4407 100644 --- a/docs/archive/architecture.pre-v1-alignment.md +++ b/docs/ver1/architecture_v1.md @@ -1,4 +1,4 @@ - # Architecture + # System Architecture (Version 1) This document describes the production architecture of the personal historical-document transcription system. The system is intentionally optimized for single-user operation, low operational overhead, and clean internal boundaries that support future growth without rewrites. @@ -265,6 +265,8 @@ Control: - first-class human review and immutable revision history +--- + ## Technology References - [FastAPI documentation](https://fastapi.tiangolo.com/) @@ -275,7 +277,14 @@ Control: ## Related Local References -- [System overview](index.md) +- [System Overview](index_v1.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- System Architecture (this document) +- [System Requirements](requirements_v1.md) +- [Data model](schema_v1.md) +- [Error Handling Policy](error_handling_v1.md) +- [Implementation Plan](implementation_plan_v1.md) ## Glossary diff --git a/docs/error_handling.md b/docs/ver1/error_handling_v1.md similarity index 96% rename from docs/error_handling.md rename to docs/ver1/error_handling_v1.md index fc7fd88..f597352 100644 --- a/docs/error_handling.md +++ b/docs/ver1/error_handling_v1.md @@ -1,4 +1,4 @@ -# Error Handling +# Error Handling Policy This document defines the canonical error-handling policy for the document transcription system. It is the single source of truth for how errors are classified, surfaced to users, logged for diagnosis, and handled across UI, API, service, worker, and provider boundaries. @@ -266,12 +266,18 @@ Change requirements: - preserve taxonomy stability; if changed, document migration impact - record noteworthy policy changes in project release notes or changelog -## Related Pages +--- -- [System overview](index.md) -- [Architecture](architecture.md) -- [Requirements](requirements.md) -- [Intent](intent.md) +## Related Local References + +- [System Overview](index_v1.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v1.md) +- [System Requirements](requirements_v1.md) +- [Data model](schema_v1.md) +- Error Handling Policy (this document) +- [Implementation Plan](implementation_plan_v1.md) ## Glossary diff --git a/docs/ver1/ver1.md b/docs/ver1/implementation_plan_v1.md similarity index 91% rename from docs/ver1/ver1.md rename to docs/ver1/implementation_plan_v1.md index ff6c21d..17a374b 100644 --- a/docs/ver1/ver1.md +++ b/docs/ver1/implementation_plan_v1.md @@ -44,7 +44,7 @@ V1 is complete when all of the following are true: 4. Freeze V1 status lifecycle to current implementation (`queued`, `processing`, `transcribed`, `failed`). ### Deliverables -- Updated `docs/schema.md` and `docs/requirements.md` traceability alignment. +- Updated `schema_v1.md` and `requirements.md` traceability alignment. - Explicit V1 data invariants section in architecture docs. ### Exit Criteria @@ -150,9 +150,8 @@ V1 is complete when all of the following are true: ### Deliverables - V1 release checklist and acceptance evidence. -- `docs/runbook.md` for incident response and operator workflows. -- `docs/migration_v1.md` for V1 migration/backfill/rollback guidance. -- `docs/release_checklist_v1.md` for release sign-off. +- `runbook_v1.md` for incident response and operator workflows. +- `release_checklist_v1.md` for release sign-off. ### Exit Criteria - Stakeholder sign-off and launch readiness achieved. @@ -187,4 +186,18 @@ A lightweight traceability table should be maintained with: - Only work required to satisfy V1 requirements enters this plan. - Nice-to-have enhancements are captured in a separate backlog document. -- Schema or contract changes after Phase 1 require explicit approval and traceability impact review. \ No newline at end of file +- Schema or contract changes after Phase 1 require explicit approval and traceability impact review. + +--- + +## Related Local References + +- [System Overview](index_v1.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v1.md) +- [System Requirements](requirements_v1.md) +- [Data model](schema_v1.md) +- [Error Handling Policy](error_handling_v1.md) +- Implementation Plan (this document) + diff --git a/docs/archive/index.pre-v1-alignment.md b/docs/ver1/index_v1.md similarity index 80% rename from docs/archive/index.pre-v1-alignment.md rename to docs/ver1/index_v1.md index ae5fcd0..4578d9d 100644 --- a/docs/archive/index.pre-v1-alignment.md +++ b/docs/ver1/index_v1.md @@ -1,10 +1,10 @@ -## Document Transcription System +## Document Transcription System Overview This project is a production application for transcribing and preserving historical family documents. It is intentionally designed for personal-scale use, with a simplicity-first architecture that is easy to operate and easy to extend. ## Start Here -Read [architecture.md](architecture.md) first. +Read [architecture_v1.md](architecture_v1.md) first. The architecture page is the primary technical reference and defines: @@ -17,7 +17,7 @@ The architecture page is the primary technical reference and defines: At a high level, users upload images or PDFs as content sources for handwritten, typed, or typeset documents, run asynchronous transcription jobs, review optional revisions, and search across accepted text. -Core capabilities: +### Core capabilities: - document grouping with one or more content sources and metadata capture - asynchronous transcription with visible job status @@ -38,16 +38,18 @@ The system runs with minimal operational overhead: This operating model keeps deployment and maintenance simple while preserving clean boundaries for future scale. +--- + ## Documentation Map -- Architecture and technical design: [architecture.md](architecture.md) -- Runtime and deployment requirements: [requirements.md](requirements.md) -- Error handling policy and operational guidance: [error_handling.md](error_handling.md) -- Domain context and transcription policy: [intent.md](intent.md) -- Transcription Methodology: [transcription_methodology.md](transcription_methodology.md) -- Data model: [schema.md](schema.md) - - +- System Overview (this document) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v1.md) +- [System Requirements](requirements_v1.md) +- [Data model](schema_v1.md) +- [Error Handling Policy](error_handling_v1.md) +- [Implementation Plan](implementation_plan_v1.md) ## Glossary diff --git a/docs/ver1/migration_v1.md b/docs/ver1/migration_v1.md deleted file mode 100644 index 55dc063..0000000 --- a/docs/ver1/migration_v1.md +++ /dev/null @@ -1,92 +0,0 @@ -# V1 Data Migration and Recovery Guidance - -This document defines migration/backfill and rollback guidance for the V1 SQLite baseline. - -## Purpose - -- provide safe procedures for local schema evolution and recovery -- reduce data-loss risk during version upgrades -- establish repeatable pre-change and post-change checks - -## Current Baseline - -- canonical relational store: SQLite -- default DB path: `./transcription.db` -- schema bootstrap may apply compatibility updates for dev/test scenarios - -## Pre-Change Checklist - -Before changing runtime version or schema behavior: - -1. Stop the app process. -2. Create a timestamped DB backup copy. -3. Capture current app commit/version. -4. Export a quick status inventory: - - job counts by status - - total documents/sources/revisions -5. Ensure sufficient disk space. - -## Backup Procedure (SQLite) - -Minimum procedure: - -1. Stop app. -2. Copy DB file to a safe location with timestamp. -3. Store backup path in release notes or change log. - -## Upgrade Procedure (V1) - -1. Perform pre-change checklist. -2. Deploy updated app version. -3. Start app and observe startup logs. -4. Verify schema bootstrap completes (if enabled). -5. Run smoke flow: - - upload valid file - - observe terminal status - - open job detail - -## Backfill Guidance - -V1 backfill is limited and conservative: - -- for records missing newly introduced non-null defaults, use explicit one-time SQL updates only after backup -- avoid destructive rewrites of `Job.text` or `Revision.text` -- never backfill by overwriting original immutable transcription output - -## Rollback Procedure - -If upgrade fails or causes data inconsistency: - -1. Stop app. -2. Restore prior DB backup file. -3. Revert app version to last known-good commit. -4. Restart app. -5. Run smoke flow and confirm stability. - -## Recovery Scenarios - -### Stale processing jobs after crash/restart - -- restart app and allow stale-job recovery to re-queue timed-out `processing` jobs -- monitor for terminal progression - -### Schema mismatch symptoms - -- errors during startup or writes indicating missing columns/indexes -- rollback to last good DB + app version -- reattempt with documented upgrade path - -## Validation Evidence - -For each upgrade rehearsal, capture: - -- backup filename/path -- pre and post job status counts -- smoke test result -- rollback rehearsal result (recommended) - -## Operational Constraints - -- treat DB backups as required before non-trivial upgrades -- do not perform in-place DB edits while app is running -- do not skip post-upgrade smoke validation diff --git a/docs/ver1/release_checklist_v1.md b/docs/ver1/release_checklist_v1.md index 61f3856..46aa174 100644 --- a/docs/ver1/release_checklist_v1.md +++ b/docs/ver1/release_checklist_v1.md @@ -18,8 +18,8 @@ Use this checklist before declaring V1 operationally complete. ## C) Operational Readiness -- [ ] `docs/runbook.md` reviewed and current. -- [ ] `docs/migration_v1.md` reviewed and current. +- [ ] `runbook_v1.md` reviewed and current. +- [ ] `migration_v1.md` reviewed and current. - [ ] Backup and rollback procedures tested at least once. - [ ] Incident escalation packet template is known to operators. @@ -28,14 +28,14 @@ Use this checklist before declaring V1 operationally complete. - [ ] Lint/type checks pass. - [ ] `pytest -m "not external" -q` passes. - [ ] Targeted external/provider checks executed (if credentials available). -- [ ] Release evidence recorded in `docs/release_evidence_v1.md`. +- [ ] Release evidence recorded in `release_evidence_v1.md`. ## E) Traceability and Documentation -- [ ] `docs/requirements.md` aligns with implemented V1 behavior. -- [ ] `docs/architecture.md`, `docs/schema.md`, and `docs/error_handling.md` are consistent. -- [ ] `docs/traceability_v1.md` is updated with current implementation and test evidence. -- [ ] `docs/ver1/ver1.md` phase status updated with evidence references. +- [ ] `requirements_v1.md` aligns with implemented V1 behavior. +- [ ] `architecture_v1.md`, `schema_v1.md`, and `error_handling_v1.md` are consistent. +- [ ] `traceability_v1.md` is updated with current implementation and test evidence. +- [ ] `implementation_plan_v1.md` phase status updated with evidence references. - [ ] REQ traceability evidence links recorded (tests/runbook/checks). ## Release Sign-Off diff --git a/docs/archive/requirements.pre-v1-alignment.md b/docs/ver1/requirements_v1.md similarity index 90% rename from docs/archive/requirements.pre-v1-alignment.md rename to docs/ver1/requirements_v1.md index 62dcf09..16a141d 100644 --- a/docs/archive/requirements.pre-v1-alignment.md +++ b/docs/ver1/requirements_v1.md @@ -1,6 +1,6 @@ ## Document Transcription System Requirements -This page captures a SysML v1.6-style requirements baseline for the production system described in [index.md](index.md). The model is represented as concise tables and traceability lists that preserve SysML-style IDs and relationship semantics. +This page captures a SysML v1.6-style requirements baseline for the production system described in [index_v1.md](index_v1.md). The model is represented as concise tables and traceability lists that preserve SysML-style IDs and relationship semantics. ## Scope @@ -77,6 +77,19 @@ This page captures a SysML v1.6-style requirements baseline for the production s - Analysis: evaluate asynchronous execution behavior and design sufficiency. - Test: automate behavioral checks through pytest suites and service-level tests. +--- + +## Related Local References + +- [System Overview](index_v1.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v1.md) +- System Requirements (this document) +- [Data model](schema_v1.md) +- [Error Handling Policy](error_handling_v1.md) +- [Implementation Plan](implementation_plan_v1.md) + ## Glossary - Document-oriented persistence: A storage approach that uses flexible document structures for variable data shapes. diff --git a/docs/runbook.md b/docs/ver1/runbook_v1.md similarity index 100% rename from docs/runbook.md rename to docs/ver1/runbook_v1.md diff --git a/docs/schema.md b/docs/ver1/schema_v1.md similarity index 85% rename from docs/schema.md rename to docs/ver1/schema_v1.md index 00bb673..618af6c 100644 --- a/docs/schema.md +++ b/docs/ver1/schema_v1.md @@ -79,6 +79,17 @@ erDiagram --- +## Related Local References + +- [System Overview](index_v1.md) +- [System Design Intent](intent.md) +- [Transcription Methodology](transcription_methodology.md) +- [System Architecture](architecture_v1.md) +- [System Requirements](requirements_v1.md) +- Data model (this document) +- [Error Handling Policy](error_handling_v1.md) +- [Implementation Plan](implementation_plan_v1.md) + ## Glossary - **Document**: logical grouping for one or more transcribed sources. diff --git a/docs/ver1/traceability_v1.md b/docs/ver1/traceability_v1.md index dc76d8b..b012db7 100644 --- a/docs/ver1/traceability_v1.md +++ b/docs/ver1/traceability_v1.md @@ -21,7 +21,7 @@ Status values: | REQ-6 | done | Background processing trigger/worker notifier and non-blocking workflow in `src/transcription/ui/components/upload.py`, `src/transcription/worker.py` | `tests/test_app.py`, `tests/services/test_workflows_reliability.py` | | REQ-7 | done | Lifespan-owned runtime resources in `src/transcription/app.py`, `src/transcription/db/runtime.py` | `tests/test_app.py`, `tests/test_db.py` | | REQ-8 | done | Centralized settings/logging initialization in `src/transcription/config.py`, `src/transcription/app.py` | `tests/test_config.py`, `tests/test_app.py` | -| REQ-9 | done | Containerized runtime baseline in `docker-compose.yml`, `Dockerfile` | `docs/release_checklist_v1.md` (Ops checklist), manual demonstration step | +| REQ-9 | done | Containerized runtime baseline in `docker-compose.yml`, `Dockerfile` | `release_checklist_v1.md` (Ops checklist), manual demonstration step | | REQ-10 | done | Explicit schema bootstrap policy + runtime controls in `src/transcription/config.py`, `src/transcription/app.py`, `src/transcription/db/operations.py` | `tests/test_db.py`, `tests/test_config.py` | | REQ-11 | done | Service/workflow persistence boundaries in `src/transcription/services/*.py`, `src/transcription/services/workflows.py` | `tests/services/test_job_service.py`, `tests/services/test_transcription_service.py` | | REQ-12 | done | Prompt artifacts in `prompts/` and loading/validation in `src/transcription/services/transcription.py` | `tests/test_prompts.py` | @@ -29,9 +29,9 @@ Status values: ## Operational Evidence (Step 3 Artifacts) -- Runbook: `docs/runbook.md` -- Migration/backfill/rollback guidance: `docs/migration_v1.md` -- Release readiness checklist: `docs/release_checklist_v1.md` +- Runbook: `runbook_v1.md` +- Migration/backfill/rollback guidance: `migration_v1.md` +- Release readiness checklist: `release_checklist_v1.md` ## Verification Cadence