#!/usr/bin/env python3 """ extractor.py — Text-Extraktion aus verschiedenen Dateiformaten. Unterstützte Formate: .txt .md → direkt lesen .docx → python-docx .pdf → pypdf (mit OCR-Fallback bei Scans) .png .jpg .jpeg → OCR via Tesseract .tif .tiff .bmp → OCR via Tesseract .webp → OCR via Tesseract Tesseract muss systemweit installiert sein: brew install tesseract tesseract-lang """ import re from pathlib import Path try: from pypdf import PdfReader _PDF_SUPPORT = True except ImportError: _PDF_SUPPORT = False try: from docx import Document as DocxDocument _DOCX_SUPPORT = True except ImportError: _DOCX_SUPPORT = False try: from PIL import Image import pytesseract _OCR_SUPPORT = True except ImportError: _OCR_SUPPORT = False try: from pdf2image import convert_from_path _PDF2IMAGE_SUPPORT = True except ImportError: _PDF2IMAGE_SUPPORT = False SUPPORTED_EXTENSIONS = { ".txt", ".md", ".pdf", ".docx", ".png", ".jpg", ".jpeg", ".tif", ".tiff", ".bmp", ".webp", } _PDF_MIN_TEXT_LENGTH = 150 _OCR_LANG = "deu+eng" def extract(path: Path) -> str: if not path.exists(): raise FileNotFoundError(f"Datei nicht gefunden: {path}") suffix = path.suffix.lower() if suffix not in SUPPORTED_EXTENSIONS: raise ValueError( f"Nicht unterstütztes Format '{suffix}'. " f"Unterstützt: {', '.join(sorted(SUPPORTED_EXTENSIONS))}" ) if suffix in (".txt", ".md"): return _read_text(path) elif suffix == ".docx": return _read_docx(path) elif suffix == ".pdf": return _read_pdf(path) else: return _ocr_image(path) def is_supported(path: Path) -> bool: return path.suffix.lower() in SUPPORTED_EXTENSIONS def _read_text(path: Path) -> str: return path.read_text(encoding="utf-8") def _read_docx(path: Path) -> str: if not _DOCX_SUPPORT: raise ImportError("python-docx nicht installiert: pip install python-docx") doc = DocxDocument(str(path)) parts = [] for element in doc.element.body: tag = element.tag.split("}")[-1] if tag == "p": runs = "".join( r.text for r in element.iter() if r.tag.endswith("}t") and r.text ) text = runs.strip() if text: parts.append(text) elif tag == "tbl": for row in element.iter(): if row.tag.endswith("}tr"): cells = [ "".join( t.text or "" for t in cell.iter() if t.tag.endswith("}t") ).strip() for cell in row if cell.tag.endswith("}tc") ] if any(cells): parts.append(" | ".join(cells)) return "\n\n".join(parts) def _read_pdf(path: Path) -> str: if not _PDF_SUPPORT: raise ImportError("pypdf nicht installiert: pip install pypdf") reader = PdfReader(str(path)) pages = [page.extract_text() or "" for page in reader.pages] text = "\n\n".join(pages).strip() if len(text) < _PDF_MIN_TEXT_LENGTH: if not _PDF2IMAGE_SUPPORT or not _OCR_SUPPORT: raise ImportError( "PDF scheint ein Scan zu sein, aber pdf2image oder pytesseract fehlen. " "Installieren: pip install pdf2image pytesseract && brew install tesseract" ) print(" ℹ️ PDF scheint ein Scan – verwende OCR-Fallback...") text = _ocr_pdf(path) return text def _ocr_pdf(path: Path) -> str: images = convert_from_path(str(path), dpi=300) pages = [] for i, img in enumerate(images, 1): page_text = pytesseract.image_to_string(img, lang=_OCR_LANG) pages.append(page_text) print(f" 📄 OCR Seite {i}/{len(images)} abgeschlossen") return _clean_ocr("\n\n".join(pages)) def _ocr_image(path: Path) -> str: if not _OCR_SUPPORT: raise ImportError( "pytesseract oder Pillow nicht installiert. " "Installieren: pip install pytesseract Pillow && brew install tesseract tesseract-lang" ) img = Image.open(str(path)) raw = pytesseract.image_to_string(img, lang=_OCR_LANG) return _clean_ocr(raw) def _clean_ocr(text: str) -> str: text = re.sub(r"[^\S\n]+", " ", text) text = re.sub(r"\n{3,}", "\n\n", text) text = re.sub(r"[|}{~`\\^]", "", text) text = re.sub(r"(\w)-\n(\w)", r"\1\2", text) return text.strip()