#!/usr/bin/env python3 """ Eurojackpot CSV Processor - Angepasst für Semikolon-getrennte CSV-Dateien Dieses Script liest zwei CSV-Dateien ein: 1. Alle möglichen Zahlenkombinationen (5 Zahlen) 2. Bereits gezogene Zahlen mit Datumsstempel Es markiert in der ersten Datei alle bereits gezogenen Kombinationen mit 1 (sonst 0). """ import pandas as pd import sys from pathlib import Path def detect_separator(filepath): """Erkennt das CSV-Trennzeichen automatisch.""" try: with open(filepath, 'r', encoding='utf-8') as f: first_line = f.readline() if ';' in first_line and first_line.count(';') > first_line.count(','): return ';' return ',' except: return ',' def load_combinations_file(filepath): """Lädt die Datei mit allen möglichen Kombinationen.""" try: sep = detect_separator(filepath) print(f"Erkanntes Trennzeichen für Kombinationen: '{sep}'") df = pd.read_csv(filepath, sep=sep) print(f"Kombinationen geladen: {len(df)} Zeilen") print(f"Spalten: {list(df.columns)}") return df except Exception as e: print(f"Fehler beim Laden der Kombinationsdatei: {e}") return None def load_drawn_numbers_file(filepath): """Lädt die Datei mit bereits gezogenen Zahlen.""" try: sep = detect_separator(filepath) print(f"Erkanntes Trennzeichen für gezogene Zahlen: '{sep}'") df = pd.read_csv(filepath, sep=sep) print(f"Gezogene Zahlen geladen: {len(df)} Zeilen") print(f"Spalten: {list(df.columns)}") return df except Exception as e: print(f"Fehler beim Laden der gezogenen Zahlen: {e}") return None def identify_number_columns(df, is_combinations=True): """Identifiziert die Spalten mit den Zahlen.""" columns = df.columns.tolist() if is_combinations: # Für Kombinationsdatei: Z1, Z2, Z3, Z4, Z5 suchen z_cols = [col for col in columns if col.startswith('Z') and len(col) == 2 and col[1:].isdigit()] z_cols = sorted(z_cols)[:5] if len(z_cols) >= 5: return z_cols # Fallback: erste 5 numerische Spalten numeric_cols = df.select_dtypes(include=['number']).columns[:5].tolist() if len(numeric_cols) >= 5: return numeric_cols # Fallback: erste 5 Spalten return columns[:5] else: # Für gezogene Zahlen: Z1-Z5 suchen (nicht SZ1, SZ2) z_cols = [col for col in columns if col.startswith('Z') and len(col) == 2 and col[1:].isdigit()] z_cols = [col for col in z_cols if not col.startswith('SZ')] # Superzahlen ausschließen z_cols = sorted(z_cols)[:5] if len(z_cols) >= 5: return z_cols # Fallback: numerische Spalten (ohne Datum) numeric_cols = df.select_dtypes(include=['number']).columns numeric_cols = [col for col in numeric_cols if 'datum' not in col.lower()][:5] if len(numeric_cols) >= 5: return numeric_cols.tolist() # Letzter Fallback return columns[:5] def create_combination_key(row, z_columns): """Erstellt einen eindeutigen Schlüssel aus den 5 Zahlen (sortiert).""" try: numbers = [int(row[col]) for col in z_columns] return tuple(sorted(numbers)) except: return None def process_eurojackpot_data(combinations_file, drawn_numbers_file, output_file): """Hauptfunktion zur Verarbeitung der Eurojackpot-Daten.""" # CSV-Dateien laden print("Lade Kombinationsdatei...") combinations_df = load_combinations_file(combinations_file) if combinations_df is None: return False print("\nLade Datei mit gezogenen Zahlen...") drawn_df = load_drawn_numbers_file(drawn_numbers_file) if drawn_df is None: return False # Spalten für Zahlen identifizieren combo_z_columns = identify_number_columns(combinations_df, is_combinations=True) drawn_z_columns = identify_number_columns(drawn_df, is_combinations=False) print(f"\nVerwendete Spalten für Kombinationen: {combo_z_columns}") print(f"Verwendete Spalten für gezogene Zahlen: {drawn_z_columns}") # Datencheck print(f"\nErste Kombination: {combinations_df[combo_z_columns].iloc[0].tolist()}") print(f"Erste gezogene Zahlen: {drawn_df[drawn_z_columns].iloc[0].tolist()}") # Set mit allen gezogenen Kombinationen erstellen print("\nErstelle Set mit gezogenen Kombinationen...") drawn_combinations = set() for _, row in drawn_df.iterrows(): combo_key = create_combination_key(row, drawn_z_columns) if combo_key: drawn_combinations.add(combo_key) print(f"Anzahl eindeutige gezogene Kombinationen: {len(drawn_combinations)}") # Beispiele anzeigen if drawn_combinations: print(f"Erste 5 gezogene Kombinationen: {list(drawn_combinations)[:5]}") # Neue Spalte für Markierungen hinzufügen (falls noch nicht vorhanden) if 'bereits_gezogen' not in combinations_df.columns: combinations_df['bereits_gezogen'] = 0 else: combinations_df['bereits_gezogen'] = 0 # Zurücksetzen print("\nMarkiere gezogene Kombinationen...") marked_count = 0 for idx, row in combinations_df.iterrows(): combo_key = create_combination_key(row, combo_z_columns) if combo_key and combo_key in drawn_combinations: combinations_df.at[idx, 'bereits_gezogen'] = 1 marked_count += 1 if marked_count <= 5: # Erste 5 Treffer anzeigen print(f"Treffer gefunden: {combo_key}") print(f"Anzahl markierte Kombinationen: {marked_count}") # Ergebnis speichern print(f"\nSpeichere Ergebnis in: {output_file}") sep = detect_separator(combinations_file) # Gleiches Trennzeichen wie Eingabe verwenden combinations_df.to_csv(output_file, sep=sep, index=False) # Statistiken ausgeben total_combinations = len(combinations_df) drawn_percentage = (marked_count / total_combinations) * 100 if total_combinations > 0 else 0 print(f"\n=== STATISTIKEN ===") print(f"Gesamte Kombinationen: {total_combinations:,}") print(f"Bereits gezogene Kombinationen: {marked_count:,}") print(f"Prozentsatz bereits gezogen: {drawn_percentage:.4f}%") print(f"Noch nicht gezogene Kombinationen: {total_combinations - marked_count:,}") return True def main(): """Hauptfunktion mit Benutzerinteraktion.""" print("=== Eurojackpot CSV Processor (Fixed) ===\n") # Dateipfade abfragen oder Standard verwenden if len(sys.argv) >= 4: combinations_file = sys.argv[1] drawn_numbers_file = sys.argv[2] output_file = sys.argv[3] else: print("Geben Sie die Dateipfade ein (oder drücken Sie Enter für Standard):") combinations_file = input("Pfad zur Kombinationsdatei: ").strip() if not combinations_file: combinations_file = "Alle_Eurojackpot_Kombinationen_mit_Status.csv" drawn_numbers_file = input("Pfad zur Datei mit gezogenen Zahlen: ").strip() if not drawn_numbers_file: drawn_numbers_file = "/Users/sebastianfrohlich/Library/Mobile Documents/com~apple~CloudDocs/Jupyter Notebooks/Eurojackpot/data/AlleEurojackpotzahlen.csv" output_file = input("Pfad für Ausgabedatei: ").strip() if not output_file: output_file = "Kombinationen_markiert_fixed.csv" # Überprüfen ob Dateien existieren if not Path(combinations_file).exists(): print(f"Fehler: Kombinationsdatei '{combinations_file}' nicht gefunden!") return if not Path(drawn_numbers_file).exists(): print(f"Fehler: Datei mit gezogenen Zahlen '{drawn_numbers_file}' nicht gefunden!") return # Verarbeitung starten success = process_eurojackpot_data(combinations_file, drawn_numbers_file, output_file) if success: print(f"\n✅ Verarbeitung erfolgreich abgeschlossen!") print(f"Ergebnis gespeichert in: {output_file}") else: print("\n❌ Fehler bei der Verarbeitung!") if __name__ == "__main__": main()