feat: extractor, normalizer, ingest v2 – OCR pipeline funktioniert

This commit is contained in:
2026-04-29 10:13:15 +02:00
parent 4fea148d61
commit 495831fb69
3 changed files with 857 additions and 0 deletions
+167
View File
@@ -0,0 +1,167 @@
#!/usr/bin/env python3
"""
extractor.py — Text-Extraktion aus verschiedenen Dateiformaten.
Unterstützte Formate:
.txt .md → direkt lesen
.docx → python-docx
.pdf → pypdf (mit OCR-Fallback bei Scans)
.png .jpg .jpeg → OCR via Tesseract
.tif .tiff .bmp → OCR via Tesseract
.webp → OCR via Tesseract
Tesseract muss systemweit installiert sein:
brew install tesseract tesseract-lang
"""
import re
from pathlib import Path
try:
from pypdf import PdfReader
_PDF_SUPPORT = True
except ImportError:
_PDF_SUPPORT = False
try:
from docx import Document as DocxDocument
_DOCX_SUPPORT = True
except ImportError:
_DOCX_SUPPORT = False
try:
from PIL import Image
import pytesseract
_OCR_SUPPORT = True
except ImportError:
_OCR_SUPPORT = False
try:
from pdf2image import convert_from_path
_PDF2IMAGE_SUPPORT = True
except ImportError:
_PDF2IMAGE_SUPPORT = False
SUPPORTED_EXTENSIONS = {
".txt", ".md",
".pdf",
".docx",
".png", ".jpg", ".jpeg", ".tif", ".tiff", ".bmp", ".webp",
}
_PDF_MIN_TEXT_LENGTH = 150
_OCR_LANG = "deu+eng"
def extract(path: Path) -> str:
if not path.exists():
raise FileNotFoundError(f"Datei nicht gefunden: {path}")
suffix = path.suffix.lower()
if suffix not in SUPPORTED_EXTENSIONS:
raise ValueError(
f"Nicht unterstütztes Format '{suffix}'. "
f"Unterstützt: {', '.join(sorted(SUPPORTED_EXTENSIONS))}"
)
if suffix in (".txt", ".md"):
return _read_text(path)
elif suffix == ".docx":
return _read_docx(path)
elif suffix == ".pdf":
return _read_pdf(path)
else:
return _ocr_image(path)
def is_supported(path: Path) -> bool:
return path.suffix.lower() in SUPPORTED_EXTENSIONS
def _read_text(path: Path) -> str:
return path.read_text(encoding="utf-8")
def _read_docx(path: Path) -> str:
if not _DOCX_SUPPORT:
raise ImportError("python-docx nicht installiert: pip install python-docx")
doc = DocxDocument(str(path))
parts = []
for element in doc.element.body:
tag = element.tag.split("}")[-1]
if tag == "p":
runs = "".join(
r.text for r in element.iter()
if r.tag.endswith("}t") and r.text
)
text = runs.strip()
if text:
parts.append(text)
elif tag == "tbl":
for row in element.iter():
if row.tag.endswith("}tr"):
cells = [
"".join(
t.text or "" for t in cell.iter()
if t.tag.endswith("}t")
).strip()
for cell in row
if cell.tag.endswith("}tc")
]
if any(cells):
parts.append(" | ".join(cells))
return "\n\n".join(parts)
def _read_pdf(path: Path) -> str:
if not _PDF_SUPPORT:
raise ImportError("pypdf nicht installiert: pip install pypdf")
reader = PdfReader(str(path))
pages = [page.extract_text() or "" for page in reader.pages]
text = "\n\n".join(pages).strip()
if len(text) < _PDF_MIN_TEXT_LENGTH:
if not _PDF2IMAGE_SUPPORT or not _OCR_SUPPORT:
raise ImportError(
"PDF scheint ein Scan zu sein, aber pdf2image oder pytesseract fehlen. "
"Installieren: pip install pdf2image pytesseract && brew install tesseract"
)
print(" ️ PDF scheint ein Scan verwende OCR-Fallback...")
text = _ocr_pdf(path)
return text
def _ocr_pdf(path: Path) -> str:
images = convert_from_path(str(path), dpi=300)
pages = []
for i, img in enumerate(images, 1):
page_text = pytesseract.image_to_string(img, lang=_OCR_LANG)
pages.append(page_text)
print(f" 📄 OCR Seite {i}/{len(images)} abgeschlossen")
return _clean_ocr("\n\n".join(pages))
def _ocr_image(path: Path) -> str:
if not _OCR_SUPPORT:
raise ImportError(
"pytesseract oder Pillow nicht installiert. "
"Installieren: pip install pytesseract Pillow && brew install tesseract tesseract-lang"
)
img = Image.open(str(path))
raw = pytesseract.image_to_string(img, lang=_OCR_LANG)
return _clean_ocr(raw)
def _clean_ocr(text: str) -> str:
text = re.sub(r"[^\S\n]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
text = re.sub(r"[|}{~`\\^]", "", text)
text = re.sub(r"(\w)-\n(\w)", r"\1\2", text)
return text.strip()