feat: extractor, normalizer, ingest v2 – OCR pipeline funktioniert
This commit is contained in:
+167
@@ -0,0 +1,167 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
extractor.py — Text-Extraktion aus verschiedenen Dateiformaten.
|
||||
|
||||
Unterstützte Formate:
|
||||
.txt .md → direkt lesen
|
||||
.docx → python-docx
|
||||
.pdf → pypdf (mit OCR-Fallback bei Scans)
|
||||
.png .jpg .jpeg → OCR via Tesseract
|
||||
.tif .tiff .bmp → OCR via Tesseract
|
||||
.webp → OCR via Tesseract
|
||||
|
||||
Tesseract muss systemweit installiert sein:
|
||||
brew install tesseract tesseract-lang
|
||||
"""
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from pypdf import PdfReader
|
||||
_PDF_SUPPORT = True
|
||||
except ImportError:
|
||||
_PDF_SUPPORT = False
|
||||
|
||||
try:
|
||||
from docx import Document as DocxDocument
|
||||
_DOCX_SUPPORT = True
|
||||
except ImportError:
|
||||
_DOCX_SUPPORT = False
|
||||
|
||||
try:
|
||||
from PIL import Image
|
||||
import pytesseract
|
||||
_OCR_SUPPORT = True
|
||||
except ImportError:
|
||||
_OCR_SUPPORT = False
|
||||
|
||||
try:
|
||||
from pdf2image import convert_from_path
|
||||
_PDF2IMAGE_SUPPORT = True
|
||||
except ImportError:
|
||||
_PDF2IMAGE_SUPPORT = False
|
||||
|
||||
|
||||
SUPPORTED_EXTENSIONS = {
|
||||
".txt", ".md",
|
||||
".pdf",
|
||||
".docx",
|
||||
".png", ".jpg", ".jpeg", ".tif", ".tiff", ".bmp", ".webp",
|
||||
}
|
||||
|
||||
_PDF_MIN_TEXT_LENGTH = 150
|
||||
_OCR_LANG = "deu+eng"
|
||||
|
||||
|
||||
def extract(path: Path) -> str:
|
||||
if not path.exists():
|
||||
raise FileNotFoundError(f"Datei nicht gefunden: {path}")
|
||||
|
||||
suffix = path.suffix.lower()
|
||||
|
||||
if suffix not in SUPPORTED_EXTENSIONS:
|
||||
raise ValueError(
|
||||
f"Nicht unterstütztes Format '{suffix}'. "
|
||||
f"Unterstützt: {', '.join(sorted(SUPPORTED_EXTENSIONS))}"
|
||||
)
|
||||
|
||||
if suffix in (".txt", ".md"):
|
||||
return _read_text(path)
|
||||
elif suffix == ".docx":
|
||||
return _read_docx(path)
|
||||
elif suffix == ".pdf":
|
||||
return _read_pdf(path)
|
||||
else:
|
||||
return _ocr_image(path)
|
||||
|
||||
|
||||
def is_supported(path: Path) -> bool:
|
||||
return path.suffix.lower() in SUPPORTED_EXTENSIONS
|
||||
|
||||
|
||||
def _read_text(path: Path) -> str:
|
||||
return path.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def _read_docx(path: Path) -> str:
|
||||
if not _DOCX_SUPPORT:
|
||||
raise ImportError("python-docx nicht installiert: pip install python-docx")
|
||||
|
||||
doc = DocxDocument(str(path))
|
||||
parts = []
|
||||
|
||||
for element in doc.element.body:
|
||||
tag = element.tag.split("}")[-1]
|
||||
if tag == "p":
|
||||
runs = "".join(
|
||||
r.text for r in element.iter()
|
||||
if r.tag.endswith("}t") and r.text
|
||||
)
|
||||
text = runs.strip()
|
||||
if text:
|
||||
parts.append(text)
|
||||
elif tag == "tbl":
|
||||
for row in element.iter():
|
||||
if row.tag.endswith("}tr"):
|
||||
cells = [
|
||||
"".join(
|
||||
t.text or "" for t in cell.iter()
|
||||
if t.tag.endswith("}t")
|
||||
).strip()
|
||||
for cell in row
|
||||
if cell.tag.endswith("}tc")
|
||||
]
|
||||
if any(cells):
|
||||
parts.append(" | ".join(cells))
|
||||
|
||||
return "\n\n".join(parts)
|
||||
|
||||
|
||||
def _read_pdf(path: Path) -> str:
|
||||
if not _PDF_SUPPORT:
|
||||
raise ImportError("pypdf nicht installiert: pip install pypdf")
|
||||
|
||||
reader = PdfReader(str(path))
|
||||
pages = [page.extract_text() or "" for page in reader.pages]
|
||||
text = "\n\n".join(pages).strip()
|
||||
|
||||
if len(text) < _PDF_MIN_TEXT_LENGTH:
|
||||
if not _PDF2IMAGE_SUPPORT or not _OCR_SUPPORT:
|
||||
raise ImportError(
|
||||
"PDF scheint ein Scan zu sein, aber pdf2image oder pytesseract fehlen. "
|
||||
"Installieren: pip install pdf2image pytesseract && brew install tesseract"
|
||||
)
|
||||
print(" ℹ️ PDF scheint ein Scan – verwende OCR-Fallback...")
|
||||
text = _ocr_pdf(path)
|
||||
|
||||
return text
|
||||
|
||||
|
||||
def _ocr_pdf(path: Path) -> str:
|
||||
images = convert_from_path(str(path), dpi=300)
|
||||
pages = []
|
||||
for i, img in enumerate(images, 1):
|
||||
page_text = pytesseract.image_to_string(img, lang=_OCR_LANG)
|
||||
pages.append(page_text)
|
||||
print(f" 📄 OCR Seite {i}/{len(images)} abgeschlossen")
|
||||
return _clean_ocr("\n\n".join(pages))
|
||||
|
||||
|
||||
def _ocr_image(path: Path) -> str:
|
||||
if not _OCR_SUPPORT:
|
||||
raise ImportError(
|
||||
"pytesseract oder Pillow nicht installiert. "
|
||||
"Installieren: pip install pytesseract Pillow && brew install tesseract tesseract-lang"
|
||||
)
|
||||
img = Image.open(str(path))
|
||||
raw = pytesseract.image_to_string(img, lang=_OCR_LANG)
|
||||
return _clean_ocr(raw)
|
||||
|
||||
|
||||
def _clean_ocr(text: str) -> str:
|
||||
text = re.sub(r"[^\S\n]+", " ", text)
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
text = re.sub(r"[|}{~`\\^]", "", text)
|
||||
text = re.sub(r"(\w)-\n(\w)", r"\1\2", text)
|
||||
return text.strip()
|
||||
Reference in New Issue
Block a user