Files
ai.bewerbung---vectorize-do…/extractor.py
T

167 lines
4.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
extractor.py — Text-Extraktion aus verschiedenen Dateiformaten.
Unterstützte Formate:
.txt .md → direkt lesen
.docx → python-docx
.pdf → pypdf (mit OCR-Fallback bei Scans)
.png .jpg .jpeg → OCR via Tesseract
.tif .tiff .bmp → OCR via Tesseract
.webp → OCR via Tesseract
Tesseract muss systemweit installiert sein:
brew install tesseract tesseract-lang
"""
import re
from pathlib import Path
try:
from pypdf import PdfReader
_PDF_SUPPORT = True
except ImportError:
_PDF_SUPPORT = False
try:
from docx import Document as DocxDocument
_DOCX_SUPPORT = True
except ImportError:
_DOCX_SUPPORT = False
try:
from PIL import Image
import pytesseract
_OCR_SUPPORT = True
except ImportError:
_OCR_SUPPORT = False
try:
from pdf2image import convert_from_path
_PDF2IMAGE_SUPPORT = True
except ImportError:
_PDF2IMAGE_SUPPORT = False
SUPPORTED_EXTENSIONS = {
".txt", ".md",
".pdf",
".docx",
".png", ".jpg", ".jpeg", ".tif", ".tiff", ".bmp", ".webp",
}
_PDF_MIN_TEXT_LENGTH = 150
_OCR_LANG = "deu+eng"
def extract(path: Path) -> str:
if not path.exists():
raise FileNotFoundError(f"Datei nicht gefunden: {path}")
suffix = path.suffix.lower()
if suffix not in SUPPORTED_EXTENSIONS:
raise ValueError(
f"Nicht unterstütztes Format '{suffix}'. "
f"Unterstützt: {', '.join(sorted(SUPPORTED_EXTENSIONS))}"
)
if suffix in (".txt", ".md"):
return _read_text(path)
elif suffix == ".docx":
return _read_docx(path)
elif suffix == ".pdf":
return _read_pdf(path)
else:
return _ocr_image(path)
def is_supported(path: Path) -> bool:
return path.suffix.lower() in SUPPORTED_EXTENSIONS
def _read_text(path: Path) -> str:
return path.read_text(encoding="utf-8")
def _read_docx(path: Path) -> str:
if not _DOCX_SUPPORT:
raise ImportError("python-docx nicht installiert: pip install python-docx")
doc = DocxDocument(str(path))
parts = []
for element in doc.element.body:
tag = element.tag.split("}")[-1]
if tag == "p":
runs = "".join(
r.text for r in element.iter()
if r.tag.endswith("}t") and r.text
)
text = runs.strip()
if text:
parts.append(text)
elif tag == "tbl":
for row in element.iter():
if row.tag.endswith("}tr"):
cells = [
"".join(
t.text or "" for t in cell.iter()
if t.tag.endswith("}t")
).strip()
for cell in row
if cell.tag.endswith("}tc")
]
if any(cells):
parts.append(" | ".join(cells))
return "\n\n".join(parts)
def _read_pdf(path: Path) -> str:
if not _PDF_SUPPORT:
raise ImportError("pypdf nicht installiert: pip install pypdf")
reader = PdfReader(str(path))
pages = [page.extract_text() or "" for page in reader.pages]
text = "\n\n".join(pages).strip()
if len(text) < _PDF_MIN_TEXT_LENGTH:
if not _PDF2IMAGE_SUPPORT or not _OCR_SUPPORT:
raise ImportError(
"PDF scheint ein Scan zu sein, aber pdf2image oder pytesseract fehlen. "
"Installieren: pip install pdf2image pytesseract && brew install tesseract"
)
print(" ️ PDF scheint ein Scan verwende OCR-Fallback...")
text = _ocr_pdf(path)
return text
def _ocr_pdf(path: Path) -> str:
images = convert_from_path(str(path), dpi=300)
pages = []
for i, img in enumerate(images, 1):
page_text = pytesseract.image_to_string(img, lang=_OCR_LANG)
pages.append(page_text)
print(f" 📄 OCR Seite {i}/{len(images)} abgeschlossen")
return _clean_ocr("\n\n".join(pages))
def _ocr_image(path: Path) -> str:
if not _OCR_SUPPORT:
raise ImportError(
"pytesseract oder Pillow nicht installiert. "
"Installieren: pip install pytesseract Pillow && brew install tesseract tesseract-lang"
)
img = Image.open(str(path))
raw = pytesseract.image_to_string(img, lang=_OCR_LANG)
return _clean_ocr(raw)
def _clean_ocr(text: str) -> str:
text = re.sub(r"[^\S\n]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
text = re.sub(r"[|}{~`\\^]", "", text)
text = re.sub(r"(\w)-\n(\w)", r"\1\2", text)
return text.strip()