generated from john/python-template
Misc updates I don't remember making
This commit is contained in:
@@ -0,0 +1,235 @@
|
||||
# extract_fs_links.py
|
||||
# Extracts FamilySearch links and IDs from PDF files.
|
||||
# 1. In FamilySearch.org: select a Person -> View tree -> Options -> Print -> Save as PDF
|
||||
# This will provide you with a tree segment that 3 levels above and one below the selected person.
|
||||
# 2. Select a person on the top row and repeat step 1. Do this for as many top-level people as needed.
|
||||
# 3. Run this script to extract the FamilySearch links and IDs from the saved PDF files.
|
||||
|
||||
import csv
|
||||
import io
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import unicodedata
|
||||
from xmlrpc import client
|
||||
from google import genai
|
||||
from google.genai import types
|
||||
import pandas as pd
|
||||
import pdfplumber
|
||||
import pypdfium2 as pdfium
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
# Configuration
|
||||
SOURCE_DIR = Path(r"C:\Proton Drive\bbchops\My files\Archive\FamilySearch")
|
||||
OUTPUT_FILE = SOURCE_DIR / "combined_pedigree_with_ids.csv"
|
||||
|
||||
FS_ID_PATTERN = re.compile(r"([A-Z0-9]{4}-[A-Z0-9]{3,4})")
|
||||
|
||||
|
||||
# --- Schema Definition for Structured Vision Extraction ---
|
||||
class PersonNode(BaseModel):
|
||||
person_name: str = Field(
|
||||
description="Full name of the person (excluding dates or badge titles)"
|
||||
)
|
||||
person_dates: str = Field(
|
||||
description="Date range (e.g. 1890-1975, 1795-Deceased, or blank)"
|
||||
)
|
||||
parent1_name: str = Field(
|
||||
default="", description="Full name of Father/Parent 1 if visible"
|
||||
)
|
||||
parent1_dates: str = Field(
|
||||
default="", description="Date range of Father/Parent 1 if visible"
|
||||
)
|
||||
parent2_name: str = Field(
|
||||
default="", description="Full name of Mother/Parent 2 if visible"
|
||||
)
|
||||
parent2_dates: str = Field(
|
||||
default="", description="Date range of Mother/Parent 2 if visible"
|
||||
)
|
||||
|
||||
|
||||
class PedigreeTree(BaseModel):
|
||||
records: list[PersonNode]
|
||||
|
||||
|
||||
# --- Helper Functions ---
|
||||
def clean_name(text: str) -> str:
|
||||
"""Normalizes string and strips dates/artifacts for reliable matching."""
|
||||
if not text:
|
||||
return ""
|
||||
text = unicodedata.normalize("NFKD", text)
|
||||
text = re.sub(r"[\u2010-\u2015\u2212\uFE58\uFE63\uFF0D]", "-", text)
|
||||
for badge in [
|
||||
"MAYFLOWER",
|
||||
"DESCENDANT",
|
||||
"PATRIOT",
|
||||
"DEAD",
|
||||
"END",
|
||||
"FamilySearch",
|
||||
]:
|
||||
text = re.sub(rf"\b{badge}\b", "", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"\b\d{4}\s*-\s*(?:\d{4}|Deceased)\b", "", text)
|
||||
return " ".join(text.split()).strip().lower()
|
||||
|
||||
|
||||
def clean_dates(text: str) -> str:
|
||||
"""Standardizes date ranges to 'YYYY-YYYY' or 'YYYY-Deceased'."""
|
||||
if not text:
|
||||
return ""
|
||||
text = unicodedata.normalize("NFKD", text)
|
||||
text = re.sub(r"[\u2010-\u2015\u2212\uFE58\uFE63\uFF0D]", "-", text)
|
||||
match = re.search(r"(\b\d{4}\s*-\s*(?:\d{4}|Deceased)\b)", text)
|
||||
return match.group(1).replace(" ", "") if match else text.strip()
|
||||
|
||||
|
||||
# --- PASS 1: Link & ID Extraction ---
|
||||
def extract_link_dataframe(source_dir: Path) -> pd.DataFrame:
|
||||
rows = []
|
||||
pdf_files = sorted(source_dir.glob("*.pdf"))
|
||||
|
||||
for pdf_path in pdf_files:
|
||||
with pdfplumber.open(str(pdf_path)) as pdf:
|
||||
for page in pdf.pages:
|
||||
hyperlinks = []
|
||||
for annot in page.hyperlinks:
|
||||
uri = annot.get("uri", "")
|
||||
match = FS_ID_PATTERN.search(uri)
|
||||
if match:
|
||||
hyperlinks.append(
|
||||
{
|
||||
"fs_id": match.group(1),
|
||||
"x0": annot.get("x0", 0),
|
||||
"top": annot.get("top", 0),
|
||||
"x1": annot.get("x1", 0),
|
||||
"bottom": annot.get("bottom", 0),
|
||||
}
|
||||
)
|
||||
|
||||
words = page.extract_words(x_tolerance=2, y_tolerance=2)
|
||||
for link in hyperlinks:
|
||||
lx_center = (link["x0"] + link["x1"]) / 2
|
||||
matched_words = [
|
||||
w
|
||||
for w in words
|
||||
if abs(((w["x0"] + w["x1"]) / 2) - lx_center) <= 28
|
||||
and (link["top"] - 5) <= w["top"] <= (link["bottom"] + 75)
|
||||
]
|
||||
matched_words.sort(key=lambda w: (w["top"], w["x0"]))
|
||||
block_text = " ".join(w["text"] for w in matched_words)
|
||||
|
||||
rows.append(
|
||||
{
|
||||
"match_key": clean_name(block_text),
|
||||
"Date_Range": clean_dates(block_text),
|
||||
"FS_ID": link["fs_id"],
|
||||
}
|
||||
)
|
||||
|
||||
df_links = pd.DataFrame(rows)
|
||||
# Deduplicate in case multiple links resolve to the same person key
|
||||
df_links = df_links.drop_duplicates(subset=["match_key"])
|
||||
return df_links
|
||||
|
||||
|
||||
# --- PASS 2: Vision OCR Pass with Gemini 2.5 Flash ---
|
||||
def extract_vision_dataframe(source_dir: Path) -> pd.DataFrame:
|
||||
client = genai.Client()
|
||||
all_nodes = []
|
||||
pdf_files = sorted(source_dir.glob("*.pdf"))
|
||||
|
||||
for pdf_path in pdf_files:
|
||||
print(f"Running Vision OCR on {pdf_path.name}...")
|
||||
# Render PDF page to high-res image in-memory
|
||||
pdf_doc = pdfium.PdfDocument(str(pdf_path))
|
||||
page = pdf_doc[0]
|
||||
image = page.render(scale=2.0).to_pil()
|
||||
|
||||
img_bytes = io.BytesIO()
|
||||
image.save(img_bytes, format="PNG")
|
||||
|
||||
prompt = (
|
||||
"Analyze this family tree chart. Extract all individuals shown as nodes along with "
|
||||
"their birth-death date ranges and their direct parents (Parent 1 and Parent 2) "
|
||||
"as indicated by the tree hierarchy lines. Strip out decorative badges like 'Mayflower Descendant'."
|
||||
)
|
||||
|
||||
response = client.models.generate_content(
|
||||
model="gemini-3.6-flash", # If unavailable on your tier, use "gemini-2.0-flash" or "gemini-1.5-flash"
|
||||
contents=[
|
||||
types.Part.from_bytes(data=img_bytes.getvalue(), mime_type="image/png"),
|
||||
prompt,
|
||||
],
|
||||
config=types.GenerateContentConfig(
|
||||
response_mime_type="application/json",
|
||||
response_schema=PedigreeTree,
|
||||
temperature=0.0,
|
||||
),
|
||||
)
|
||||
|
||||
tree_data: PedigreeTree = response.parsed
|
||||
for item in tree_data.records:
|
||||
all_nodes.append(
|
||||
{
|
||||
"Person": item.person_name.strip(),
|
||||
"Person_Dates": clean_dates(item.person_dates),
|
||||
"Parent1": item.parent1_name.strip(),
|
||||
"Parent1_Dates": clean_dates(item.parent1_dates),
|
||||
"Parent2": item.parent2_name.strip(),
|
||||
"Parent2_Dates": clean_dates(item.parent2_dates),
|
||||
}
|
||||
)
|
||||
|
||||
return pd.DataFrame(all_nodes).drop_duplicates(subset=["Person", "Person_Dates"])
|
||||
|
||||
|
||||
# --- PASS 3: Merging and Populating IDs ---
|
||||
def build_combined_pedigree():
|
||||
print("Pass 1: Extracting URLs and bounding boxes...")
|
||||
df_links = extract_link_dataframe(SOURCE_DIR)
|
||||
|
||||
print("Pass 2: Running Gemini Vision OCR on pedigree charts...")
|
||||
df_pedigree = extract_vision_dataframe(SOURCE_DIR)
|
||||
|
||||
print("Pass 3: Joining IDs into Pedigree structure...")
|
||||
# Lookup dictionary mapping cleaned name -> FS ID
|
||||
id_lookup = dict(zip(df_links["match_key"], df_links["FS_ID"]))
|
||||
|
||||
def get_id(name):
|
||||
return id_lookup.get(clean_name(name), "")
|
||||
|
||||
df_pedigree["Person_FS_ID"] = df_pedigree["Person"].apply(get_id)
|
||||
df_pedigree["Parent1_FS_ID"] = df_pedigree["Parent1"].apply(get_id)
|
||||
df_pedigree["Parent2_FS_ID"] = df_pedigree["Parent2"].apply(get_id)
|
||||
|
||||
# Reorder into the requested final column layout
|
||||
final_df = df_pedigree[
|
||||
[
|
||||
"Person",
|
||||
"Person_Dates",
|
||||
"Person_FS_ID",
|
||||
"Parent1",
|
||||
"Parent1_Dates",
|
||||
"Parent1_FS_ID",
|
||||
"Parent2",
|
||||
"Parent2_Dates",
|
||||
"Parent2_FS_ID",
|
||||
]
|
||||
].rename(
|
||||
columns={
|
||||
"Person_Dates": "Date Range",
|
||||
"Person_FS_ID": "FS ID",
|
||||
"Parent1": "Parent 1",
|
||||
"Parent1_Dates": "Date Range",
|
||||
"Parent1_FS_ID": "FS ID",
|
||||
"Parent2": "Parent 2",
|
||||
"Parent2_Dates": "Date Range",
|
||||
"Parent2_FS_ID": "FS ID",
|
||||
}
|
||||
)
|
||||
|
||||
final_df.to_csv(OUTPUT_FILE, index=False, encoding="utf-8-sig")
|
||||
print(f"\nCompleted! Combined table exported to:\n{OUTPUT_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
build_combined_pedigree()
|
||||
Reference in New Issue
Block a user