{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":106680,"databundleVersionId":13374319,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Need for a uniform interface of running models\n\nAs described in the official competition page, to win the prize money, a prerequisite is that the code has to be made open-source. In addition, the top 10 submissions/teams will be invited to become co-authors in a scientific paper that involves further stress-testing of their models in a subsequent phase with many other datasets outside Kaggle platform. **To enable such further analyses and re-use of the models by the community, we strongly encourage** the participants to adhere to a code template that we provide through this repository that enables a uniform interface of running models: [https://github.com/uio-bmi/predict-airr](https://github.com/uio-bmi/predict-airr)\n\n\nIdeally, all the methods can be run in a unified way, e.g.,\n\n`python3 -m submission.main --train_dir /path/to/train_dir --test_dirs /path/to/test_dir_1 /path/to/test_dir_2 --out_dir /path/to/output_dir --n_jobs 4 --device cpu`\n\n## Adhering to code template on Kaggle Notebooks\n\nThose participants who make use of Kaggle resources and Kaggle notebooks to develop and run their code are also strongly encouraged to copy the code template, particularly the `ImmuneStatePredictor` class and any utility functions from the provided code template repository and adhere to the code template to enable a unified way of running different methods at a later stage. In this notebook, we copied the code template below for participants to paste into their respective Kaggle notebooks and edit as needed.","metadata":{}},{"cell_type":"code","source":"## imports required for the basic code template below.\n\nimport os\nfrom tqdm import tqdm\nimport pandas as pd\nimport numpy as np\nimport torch\nimport glob\nimport sys\nimport argparse\nfrom collections import defaultdict\nfrom typing import Iterator, Tuple, Union, List\n\ntrain_datasets_dir = \"/kaggle/input/adaptive-immune-profiling-challenge-2025/train_datasets/train_datasets\"\ntest_datasets_dir = \"/kaggle/input/adaptive-immune-profiling-challenge-2025/test_datasets/test_datasets\"\nresults_dir = \"/kaggle/working/results\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T05:51:33.864456Z","iopub.execute_input":"2025-11-30T05:51:33.864822Z","iopub.status.idle":"2025-11-30T05:51:38.815723Z","shell.execute_reply.started":"2025-11-30T05:51:33.864798Z","shell.execute_reply":"2025-11-30T05:51:38.814590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## some utility functions such as data loaders, etc.\n\ndef load_data_generator(data_dir: str, metadata_filename='metadata.csv') -> Iterator[\n    Union[Tuple[str, pd.DataFrame, bool], Tuple[str, pd.DataFrame]]]:\n    \"\"\"\n    A generator to load immune repertoire data.\n\n    This function operates in two modes:\n    1.  If metadata is found, it yields data based on the metadata file.\n    2.  If metadata is NOT found, it uses glob to find and yield all '.tsv'\n        files in the directory.\n\n    Args:\n        data_dir (str): The path to the directory containing the data.\n\n    Yields:\n        An iterator of tuples. The format depends on the mode:\n        - With metadata: (repertoire_id, pd.DataFrame, label_positive)\n        - Without metadata: (filename, pd.DataFrame)\n    \"\"\"\n    metadata_path = os.path.join(data_dir, metadata_filename)\n\n    if os.path.exists(metadata_path):\n        metadata_df = pd.read_csv(metadata_path)\n        for row in metadata_df.itertuples(index=False):\n            file_path = os.path.join(data_dir, row.filename)\n            try:\n                repertoire_df = pd.read_csv(file_path, sep='\\t')\n                yield row.repertoire_id, repertoire_df, row.label_positive\n            except FileNotFoundError:\n                print(f\"Warning: File '{row.filename}' listed in metadata not found. Skipping.\")\n                continue\n    else:\n        search_pattern = os.path.join(data_dir, '*.tsv')\n        tsv_files = glob.glob(search_pattern)\n        for file_path in sorted(tsv_files):\n            try:\n                filename = os.path.basename(file_path)\n                repertoire_df = pd.read_csv(file_path, sep='\\t')\n                yield filename, repertoire_df\n            except Exception as e:\n                print(f\"Warning: Could not read file '{file_path}'. Error: {e}. Skipping.\")\n                continue\n\n\ndef load_full_dataset(data_dir: str) -> pd.DataFrame:\n    \"\"\"\n    Loads all TSV files from a directory and concatenates them into a single DataFrame.\n\n    This function handles two scenarios:\n    1. If metadata.csv exists, it loads data based on the metadata and adds\n       'repertoire_id' and 'label_positive' columns.\n    2. If metadata.csv does not exist, it loads all .tsv files and adds\n       a 'filename' column as an identifier.\n\n    Args:\n        data_dir (str): The path to the data directory.\n\n    Returns:\n        pd.DataFrame: A single, concatenated DataFrame containing all the data.\n    \"\"\"\n    metadata_path = os.path.join(data_dir, 'metadata.csv')\n    df_list = []\n    data_loader = load_data_generator(data_dir=data_dir)\n\n    if os.path.exists(metadata_path):\n        metadata_df = pd.read_csv(metadata_path)\n        total_files = len(metadata_df)\n        for rep_id, data_df, label in tqdm(data_loader, total=total_files, desc=\"Loading files\"):\n            data_df['ID'] = rep_id\n            data_df['label_positive'] = label\n            df_list.append(data_df)\n    else:\n        search_pattern = os.path.join(data_dir, '*.tsv')\n        total_files = len(glob.glob(search_pattern))\n        for filename, data_df in tqdm(data_loader, total=total_files, desc=\"Loading files\"):\n            data_df['ID'] = os.path.basename(filename).replace(\".tsv\", \"\")\n            df_list.append(data_df)\n\n    if not df_list:\n        print(\"Warning: No data files were loaded.\")\n        return pd.DataFrame()\n\n    full_dataset_df = pd.concat(df_list, ignore_index=True)\n    return full_dataset_df\n\n\ndef load_and_encode_kmers(data_dir: str, k: int = 3) -> Tuple[pd.DataFrame, pd.DataFrame]:\n    \"\"\"\n    Loading and k-mer encoding of repertoire data.\n\n    Args:\n        data_dir: Path to data directory\n        k: K-mer length\n\n    Returns:\n        Tuple of (encoded_features_df, metadata_df)\n        metadata_df always contains 'ID', and 'label_positive' if available\n    \"\"\"\n    from collections import Counter\n\n    metadata_path = os.path.join(data_dir, 'metadata.csv')\n    data_loader = load_data_generator(data_dir=data_dir)\n\n    repertoire_features = []\n    metadata_records = []\n\n    search_pattern = os.path.join(data_dir, '*.tsv')\n    total_files = len(glob.glob(search_pattern))\n\n    for item in tqdm(data_loader, total=total_files, desc=f\"Encoding {k}-mers\"):\n        if os.path.exists(metadata_path):\n            rep_id, data_df, label = item\n        else:\n            filename, data_df = item\n            rep_id = os.path.basename(filename).replace(\".tsv\", \"\")\n            label = None\n\n        kmer_counts = Counter()\n        for seq in data_df['junction_aa'].dropna():\n            for i in range(len(seq) - k + 1):\n                kmer_counts[seq[i:i + k]] += 1\n\n        repertoire_features.append({\n            'ID': rep_id,\n            **kmer_counts\n        })\n\n        metadata_record = {'ID': rep_id}\n        if label is not None:\n            metadata_record['label_positive'] = label\n        metadata_records.append(metadata_record)\n\n        del data_df, kmer_counts\n\n    features_df = pd.DataFrame(repertoire_features).fillna(0).set_index('ID')\n    features_df.fillna(0)\n    metadata_df = pd.DataFrame(metadata_records)\n\n    return features_df, metadata_df\n\n\ndef save_tsv(df: pd.DataFrame, path: str):\n    os.makedirs(os.path.dirname(path), exist_ok=True)\n    df.to_csv(path, sep='\\t', index=False)\n\n\ndef get_repertoire_ids(data_dir: str) -> list:\n    \"\"\"\n    Retrieves repertoire IDs from the metadata file or filenames in the directory.\n\n    Args:\n        data_dir (str): The path to the data directory.\n\n    Returns:\n        list: A list of repertoire IDs.\n    \"\"\"\n    metadata_path = os.path.join(data_dir, 'metadata.csv')\n\n    if os.path.exists(metadata_path):\n        metadata_df = pd.read_csv(metadata_path)\n        repertoire_ids = metadata_df['repertoire_id'].tolist()\n    else:\n        search_pattern = os.path.join(data_dir, '*.tsv')\n        tsv_files = glob.glob(search_pattern)\n        repertoire_ids = [os.path.basename(f).replace('.tsv', '') for f in sorted(tsv_files)]\n\n    return repertoire_ids\n\n\ndef generate_random_top_sequences_df(n_seq: int = 50000) -> pd.DataFrame:\n    \"\"\"\n    Generates a random DataFrame simulating top important sequences.\n\n    Args:\n        n_seq (int): Number of sequences to generate.\n\n    Returns:\n        pd.DataFrame: A DataFrame with columns 'ID', 'dataset', 'junction_aa', 'v_call', 'j_call'.\n    \"\"\"\n    seqs = set()\n    while len(seqs) < n_seq:\n        seq = ''.join(np.random.choice(list('ACDEFGHIKLMNPQRSTVWY'), size=15))\n        seqs.add(seq)\n    data = {\n        'junction_aa': list(seqs),\n        'v_call': ['TRBV20-1'] * n_seq,\n        'j_call': ['TRBJ2-7'] * n_seq,\n        'importance_score': np.random.rand(n_seq)\n    }\n    return pd.DataFrame(data)\n\n\ndef validate_dirs_and_files(train_dir: str, test_dirs: List[str], out_dir: str) -> None:\n    assert os.path.isdir(train_dir), f\"Train directory `{train_dir}` does not exist.\"\n    train_tsvs = glob.glob(os.path.join(train_dir, \"*.tsv\"))\n    assert train_tsvs, f\"No .tsv files found in train directory `{train_dir}`.\"\n    metadata_path = os.path.join(train_dir, \"metadata.csv\")\n    assert os.path.isfile(metadata_path), f\"`metadata.csv` not found in train directory `{train_dir}`.\"\n\n    for test_dir in test_dirs:\n        assert os.path.isdir(test_dir), f\"Test directory `{test_dir}` does not exist.\"\n        test_tsvs = glob.glob(os.path.join(test_dir, \"*.tsv\"))\n        assert test_tsvs, f\"No .tsv files found in test directory `{test_dir}`.\"\n\n    try:\n        os.makedirs(out_dir, exist_ok=True)\n        test_file = os.path.join(out_dir, \"test_write_permission.tmp\")\n        with open(test_file, \"w\") as f:\n            f.write(\"test\")\n        os.remove(test_file)\n    except Exception as e:\n        print(f\"Failed to create or write to output directory `{out_dir}`: {e}\")\n        sys.exit(1)\n\n\ndef concatenate_output_files(out_dir: str) -> None:\n    \"\"\"\n    Concatenates all test predictions and important sequences TSV files from the output directory.\n\n    This function finds all files matching the patterns:\n    - *_test_predictions.tsv\n    - *_important_sequences.tsv\n\n    and concatenates them to match the expected output format of submissions.csv.\n\n    Args:\n        out_dir (str): Path to the output directory containing the TSV files.\n\n    Returns:\n        pd.DataFrame: Concatenated DataFrame with predictions followed by important sequences.\n                     Columns: ['ID', 'dataset', 'label_positive_probability', 'junction_aa', 'v_call', 'j_call']\n    \"\"\"\n    predictions_pattern = os.path.join(out_dir, '*_test_predictions.tsv')\n    sequences_pattern = os.path.join(out_dir, '*_important_sequences.tsv')\n\n    predictions_files = sorted(glob.glob(predictions_pattern))\n    sequences_files = sorted(glob.glob(sequences_pattern))\n\n    df_list = []\n\n    for pred_file in predictions_files:\n        try:\n            df = pd.read_csv(pred_file, sep='\\t')\n            df_list.append(df)\n        except Exception as e:\n            print(f\"Warning: Could not read predictions file '{pred_file}'. Error: {e}. Skipping.\")\n            continue\n\n    for seq_file in sequences_files:\n        try:\n            df = pd.read_csv(seq_file, sep='\\t')\n            df_list.append(df)\n        except Exception as e:\n            print(f\"Warning: Could not read sequences file '{seq_file}'. Error: {e}. Skipping.\")\n            continue\n\n    if not df_list:\n        print(\"Warning: No output files were found to concatenate.\")\n        concatenated_df = pd.DataFrame(\n            columns=['ID', 'dataset', 'label_positive_probability', 'junction_aa', 'v_call', 'j_call'])\n    else:\n        concatenated_df = pd.concat(df_list, ignore_index=True)\n    submissions_file = os.path.join(out_dir, 'submissions.csv')\n    concatenated_df.to_csv(submissions_file, index=False)\n    print(f\"Concatenated output written to `{submissions_file}`.\")\n\n\ndef get_dataset_pairs(train_dir: str, test_dir: str) -> List[Tuple[str, List[str]]]:\n    \"\"\"Returns list of (train_path, [test_paths]) tuples for dataset pairs.\"\"\"\n    test_groups = defaultdict(list)\n    for test_name in sorted(os.listdir(test_dir)):\n        if test_name.startswith(\"test_dataset_\"):\n            base_id = test_name.replace(\"test_dataset_\", \"\").split(\"_\")[0]\n            test_groups[base_id].append(os.path.join(test_dir, test_name))\n\n    pairs = []\n    for train_name in sorted(os.listdir(train_dir)):\n        if train_name.startswith(\"train_dataset_\"):\n            train_id = train_name.replace(\"train_dataset_\", \"\")\n            train_path = os.path.join(train_dir, train_name)\n            pairs.append((train_path, test_groups.get(train_id, [])))\n\n    return pairs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T05:51:45.508506Z","iopub.execute_input":"2025-11-30T05:51:45.509019Z","iopub.status.idle":"2025-11-30T05:51:45.542594Z","shell.execute_reply.started":"2025-11-30T05:51:45.508991Z","shell.execute_reply":"2025-11-30T05:51:45.541597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Explorative Datenanalyse (EDA) für Adaptive Immune Profiling Challenge\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import Counter\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Setze Stil für bessere Visualisierungen\nsns.set_style(\"whitegrid\")\nplt.rcParams['figure.figsize'] = (12, 6)\n\n## 1. Überblick über die Datenstruktur\n\ndef get_data_overview(train_dir, test_dir):\n    \"\"\"Gibt einen Überblick über die Datenstruktur\"\"\"\n    \n    print(\"=\" * 80)\n    print(\"DATENSTRUKTUR ÜBERBLICK\")\n    print(\"=\" * 80)\n    \n    # Training Datasets\n    train_datasets = [d for d in os.listdir(train_dir) if d.startswith('train_dataset_')]\n    print(f\"\\n📁 Anzahl Training Datasets: {len(train_datasets)}\")\n    print(f\"Training Datasets: {train_datasets[:5]}...\" if len(train_datasets) > 5 else f\"Training Datasets: {train_datasets}\")\n    \n    # Test Datasets\n    test_datasets = [d for d in os.listdir(test_dir) if d.startswith('test_dataset_')]\n    print(f\"\\n📁 Anzahl Test Datasets: {len(test_datasets)}\")\n    print(f\"Test Datasets: {test_datasets[:5]}...\" if len(test_datasets) > 5 else f\"Test Datasets: {test_datasets}\")\n    \n    return train_datasets, test_datasets\n\n\n## 2. Analyse eines einzelnen Datensatzes\n\ndef analyze_single_dataset(dataset_path, dataset_name):\n    \"\"\"Detaillierte Analyse eines einzelnen Datensatzes\"\"\"\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(f\"ANALYSE: {dataset_name}\")\n    print(\"=\" * 80)\n    \n    # Lade Metadata\n    metadata_path = os.path.join(dataset_path, 'metadata.csv')\n    if os.path.exists(metadata_path):\n        metadata = pd.read_csv(metadata_path)\n        print(f\"\\n📊 Metadata Info:\")\n        print(f\"   - Anzahl Repertoires: {len(metadata)}\")\n        print(f\"   - Spalten: {list(metadata.columns)}\")\n        \n        if 'label_positive' in metadata.columns:\n            label_dist = metadata['label_positive'].value_counts()\n            print(f\"\\n🏷️  Label-Verteilung:\")\n            print(f\"   - Positive (1): {label_dist.get(1, 0)} ({label_dist.get(1, 0)/len(metadata)*100:.1f}%)\")\n            print(f\"   - Negative (0): {label_dist.get(0, 0)} ({label_dist.get(0, 0)/len(metadata)*100:.1f}%)\")\n            \n            # Visualisierung\n            fig, ax = plt.subplots(1, 2, figsize=(12, 4))\n            \n            # Label Distribution\n            label_dist.plot(kind='bar', ax=ax[0], color=['#ff6b6b', '#4ecdc4'])\n            ax[0].set_title(f'Label-Verteilung - {dataset_name}')\n            ax[0].set_xlabel('Label')\n            ax[0].set_ylabel('Anzahl')\n            ax[0].set_xticklabels(['Negativ (0)', 'Positiv (1)'], rotation=0)\n            \n            # Pie Chart\n            ax[1].pie(label_dist.values, labels=['Negativ', 'Positiv'], \n                     autopct='%1.1f%%', colors=['#ff6b6b', '#4ecdc4'])\n            ax[1].set_title('Label-Proportion')\n            \n            plt.tight_layout()\n            plt.show()\n    \n    # Lade ein Beispiel-Repertoire\n    tsv_files = glob.glob(os.path.join(dataset_path, '*.tsv'))\n    if tsv_files:\n        print(f\"\\n📄 Anzahl TSV-Dateien: {len(tsv_files)}\")\n        \n        # Lade erstes Repertoire\n        sample_rep = pd.read_csv(tsv_files[0], sep='\\t')\n        print(f\"\\n🔬 Beispiel-Repertoire: {os.path.basename(tsv_files[0])}\")\n        print(f\"   - Anzahl Sequenzen: {len(sample_rep)}\")\n        print(f\"   - Spalten: {list(sample_rep.columns)}\")\n        print(f\"\\n   Erste Zeilen:\")\n        print(sample_rep.head(3))\n        \n        return metadata, sample_rep\n    \n    return metadata, None\n\n\n\n## 3. CDR3-Sequenz Analyse\n\ndef analyze_cdr3_sequences(dataset_path, sample_size=5):\n    \"\"\"Analysiert CDR3-Sequenzen (junction_aa)\"\"\"\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(\"CDR3-SEQUENZ ANALYSE\")\n    print(\"=\" * 80)\n    \n    # Sammle Daten aus mehreren Repertoires\n    all_sequences = []\n    all_lengths = []\n    all_v_genes = []\n    all_j_genes = []\n    \n    tsv_files = glob.glob(os.path.join(dataset_path, '*.tsv'))\n    \n    print(f\"\\n⏳ Analysiere {min(sample_size, len(tsv_files))} Repertoires...\")\n    \n    for tsv_file in tsv_files[:sample_size]:\n        df = pd.read_csv(tsv_file, sep='\\t')\n        \n        if 'junction_aa' in df.columns:\n            sequences = df['junction_aa'].dropna()\n            all_sequences.extend(sequences.tolist())\n            all_lengths.extend(sequences.str.len().tolist())\n        \n        if 'v_call' in df.columns:\n            all_v_genes.extend(df['v_call'].dropna().tolist())\n        \n        if 'j_call' in df.columns:\n            all_j_genes.extend(df['j_call'].dropna().tolist())\n    \n    print(f\"\\n📊 Statistiken über {len(all_sequences)} Sequenzen:\")\n    print(f\"   - Mittlere Länge: {np.mean(all_lengths):.2f} ± {np.std(all_lengths):.2f}\")\n    print(f\"   - Min Länge: {np.min(all_lengths)}\")\n    print(f\"   - Max Länge: {np.max(all_lengths)}\")\n    print(f\"   - Median Länge: {np.median(all_lengths):.0f}\")\n    \n    # Visualisierungen\n    fig, axes = plt.subplots(2, 2, figsize=(15, 10))\n    \n    # 1. Längenverteilung\n    axes[0, 0].hist(all_lengths, bins=50, color='skyblue', edgecolor='black', alpha=0.7)\n    axes[0, 0].axvline(np.mean(all_lengths), color='red', linestyle='--', label=f'Mean: {np.mean(all_lengths):.1f}')\n    axes[0, 0].axvline(np.median(all_lengths), color='green', linestyle='--', label=f'Median: {np.median(all_lengths):.1f}')\n    axes[0, 0].set_xlabel('CDR3 Länge (Aminosäuren)')\n    axes[0, 0].set_ylabel('Häufigkeit')\n    axes[0, 0].set_title('Verteilung der CDR3-Längen')\n    axes[0, 0].legend()\n    \n    # 2. Aminosäure-Komposition\n    aa_counter = Counter()\n    for seq in all_sequences[:10000]:  # Sample für Performance\n        aa_counter.update(seq)\n    \n    aa_df = pd.DataFrame(aa_counter.most_common(20), columns=['Aminosäure', 'Häufigkeit'])\n    axes[0, 1].bar(aa_df['Aminosäure'], aa_df['Häufigkeit'], color='coral')\n    axes[0, 1].set_xlabel('Aminosäure')\n    axes[0, 1].set_ylabel('Häufigkeit')\n    axes[0, 1].set_title('Top 20 Aminosäuren in CDR3-Sequenzen')\n    axes[0, 1].tick_params(axis='x', rotation=45)\n    \n    # 3. V-Gene Verteilung\n    if all_v_genes:\n        v_counter = Counter(all_v_genes)\n        v_top = pd.DataFrame(v_counter.most_common(15), columns=['V-Gen', 'Häufigkeit'])\n        axes[1, 0].barh(v_top['V-Gen'], v_top['Häufigkeit'], color='lightgreen')\n        axes[1, 0].set_xlabel('Häufigkeit')\n        axes[1, 0].set_title('Top 15 V-Gene')\n        axes[1, 0].invert_yaxis()\n    \n    # 4. J-Gene Verteilung\n    if all_j_genes:\n        j_counter = Counter(all_j_genes)\n        j_top = pd.DataFrame(j_counter.most_common(15), columns=['J-Gen', 'Häufigkeit'])\n        axes[1, 1].barh(j_top['J-Gen'], j_top['Häufigkeit'], color='plum')\n        axes[1, 1].set_xlabel('Häufigkeit')\n        axes[1, 1].set_title('Top 15 J-Gene')\n        axes[1, 1].invert_yaxis()\n    \n    plt.tight_layout()\n    plt.show()\n    \n    return all_sequences, all_lengths, all_v_genes, all_j_genes\n\n## 4. Repertoire-Level Analyse\n\ndef analyze_repertoire_characteristics(dataset_path, sample_size=20):\n    \"\"\"Analysiert Charakteristiken auf Repertoire-Ebene\"\"\"\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(\"REPERTOIRE-LEVEL ANALYSE\")\n    print(\"=\" * 80)\n    \n    metadata_path = os.path.join(dataset_path, 'metadata.csv')\n    metadata = pd.read_csv(metadata_path)\n    \n    repertoire_stats = []\n    \n    print(f\"\\n⏳ Analysiere {min(sample_size, len(metadata))} Repertoires...\")\n    \n    for idx, row in metadata.head(sample_size).iterrows():\n        file_path = os.path.join(dataset_path, row['filename'])\n        df = pd.read_csv(file_path, sep='\\t')\n        \n        stats = {\n            'repertoire_id': row['repertoire_id'],\n            'label': row.get('label_positive', None),\n            'n_sequences': len(df),\n            'mean_cdr3_length': df['junction_aa'].str.len().mean() if 'junction_aa' in df.columns else None,\n            'unique_v_genes': df['v_call'].nunique() if 'v_call' in df.columns else None,\n            'unique_j_genes': df['j_call'].nunique() if 'j_call' in df.columns else None,\n        }\n        repertoire_stats.append(stats)\n    \n    rep_df = pd.DataFrame(repertoire_stats)\n    \n    print(\"\\n📊 Repertoire-Statistiken:\")\n    print(rep_df.describe())\n    \n    # Visualisierung\n    if 'label' in rep_df.columns and rep_df['label'].notna().any():\n        fig, axes = plt.subplots(2, 2, figsize=(15, 10))\n        \n        # 1. Anzahl Sequenzen pro Label\n        rep_df.boxplot(column='n_sequences', by='label', ax=axes[0, 0])\n        axes[0, 0].set_title('Anzahl Sequenzen pro Repertoire')\n        axes[0, 0].set_xlabel('Label')\n        axes[0, 0].set_ylabel('Anzahl Sequenzen')\n        plt.sca(axes[0, 0])\n        plt.xticks([1, 2], ['Negativ (0)', 'Positiv (1)'])\n        \n        # 2. Mittlere CDR3-Länge pro Label\n        rep_df.boxplot(column='mean_cdr3_length', by='label', ax=axes[0, 1])\n        axes[0, 1].set_title('Mittlere CDR3-Länge pro Repertoire')\n        axes[0, 1].set_xlabel('Label')\n        axes[0, 1].set_ylabel('Mittlere Länge')\n        plt.sca(axes[0, 1])\n        plt.xticks([1, 2], ['Negativ (0)', 'Positiv (1)'])\n        \n        # 3. Unique V-Gene pro Label\n        rep_df.boxplot(column='unique_v_genes', by='label', ax=axes[1, 0])\n        axes[1, 0].set_title('Anzahl einzigartiger V-Gene')\n        axes[1, 0].set_xlabel('Label')\n        axes[1, 0].set_ylabel('Anzahl V-Gene')\n        plt.sca(axes[1, 0])\n        plt.xticks([1, 2], ['Negativ (0)', 'Positiv (1)'])\n        \n        # 4. Unique J-Gene pro Label\n        rep_df.boxplot(column='unique_j_genes', by='label', ax=axes[1, 1])\n        axes[1, 1].set_title('Anzahl einzigartiger J-Gene')\n        axes[1, 1].set_xlabel('Label')\n        axes[1, 1].set_ylabel('Anzahl J-Gene')\n        plt.sca(axes[1, 1])\n        plt.xticks([1, 2], ['Negativ (0)', 'Positiv (1)'])\n        \n        plt.tight_layout()\n        plt.show()\n    \n    return rep_df\n\n\n\n## 5. K-mer Analyse\n\ndef analyze_kmers(sequences, k=3, top_n=20):\n    \"\"\"Analysiert K-mer Häufigkeiten\"\"\"\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(f\"{k}-MER ANALYSE\")\n    print(\"=\" * 80)\n    \n    kmer_counter = Counter()\n    \n    print(f\"\\n⏳ Extrahiere {k}-mere aus {len(sequences)} Sequenzen...\")\n    \n    for seq in sequences[:10000]:  # Sample für Performance\n        if isinstance(seq, str):\n            for i in range(len(seq) - k + 1):\n                kmer_counter[seq[i:i+k]] += 1\n    \n    print(f\"\\n📊 Gefundene einzigartige {k}-mere: {len(kmer_counter)}\")\n    \n    # Top K-mere\n    top_kmers = pd.DataFrame(kmer_counter.most_common(top_n), \n                             columns=[f'{k}-mer', 'Häufigkeit'])\n    \n    print(f\"\\nTop {top_n} {k}-mere:\")\n    print(top_kmers)\n    \n    # Visualisierung\n    plt.figure(figsize=(12, 6))\n    plt.bar(range(len(top_kmers)), top_kmers['Häufigkeit'], color='steelblue')\n    plt.xticks(range(len(top_kmers)), top_kmers[f'{k}-mer'], rotation=45, ha='right')\n    plt.xlabel(f'{k}-mer')\n    plt.ylabel('Häufigkeit')\n    plt.title(f'Top {top_n} häufigste {k}-mere')\n    plt.tight_layout()\n    plt.show()\n    \n    return kmer_counter\n\n## 6. Vergleich zwischen Training und Test Datasets\n\ndef compare_train_test_datasets(train_dir, test_dir, dataset_id):\n    \"\"\"Vergleicht korrespondierende Train/Test Datasets\"\"\"\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(f\"TRAIN vs TEST VERGLEICH - Dataset {dataset_id}\")\n    print(\"=\" * 80)\n    \n    train_path = os.path.join(train_dir, f'train_dataset_{dataset_id}')\n    test_paths = glob.glob(os.path.join(test_dir, f'test_dataset_{dataset_id}*'))\n    \n    if not os.path.exists(train_path) or not test_paths:\n        print(\"⚠️  Korrespondierende Datasets nicht gefunden\")\n        return\n    \n    # Training Data\n    train_metadata = pd.read_csv(os.path.join(train_path, 'metadata.csv'))\n    train_tsv_count = len(glob.glob(os.path.join(train_path, '*.tsv')))\n    \n    print(f\"\\n🎓 Training Dataset:\")\n    print(f\"   - Repertoires: {len(train_metadata)}\")\n    print(f\"   - TSV Files: {train_tsv_count}\")\n    if 'label_positive' in train_metadata.columns:\n        print(f\"   - Positive: {train_metadata['label_positive'].sum()}\")\n        print(f\"   - Negative: {(train_metadata['label_positive'] == 0).sum()}\")\n    \n    # Test Data\n    print(f\"\\n🧪 Test Dataset(s):\")\n    for test_path in test_paths:\n        test_name = os.path.basename(test_path)\n        test_tsv_count = len(glob.glob(os.path.join(test_path, '*.tsv')))\n        print(f\"   - {test_name}: {test_tsv_count} Repertoires\")\n\n## 7. Zusammenfassung aller Datasets\n\ndef summarize_all_datasets(train_dir):\n    \"\"\"Erstellt eine Zusammenfassung aller Datasets\"\"\"\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(\"ZUSAMMENFASSUNG ALLER DATASETS\")\n    print(\"=\" * 80)\n    \n    summary_data = []\n    \n    train_datasets = [d for d in os.listdir(train_dir) if d.startswith('train_dataset_')]\n    \n    for dataset_name in tqdm(train_datasets, desc=\"Analysiere Datasets\"):\n        dataset_path = os.path.join(train_dir, dataset_name)\n        metadata_path = os.path.join(dataset_path, 'metadata.csv')\n        \n        if os.path.exists(metadata_path):\n            metadata = pd.read_csv(metadata_path)\n            \n            summary = {\n                'dataset': dataset_name,\n                'n_repertoires': len(metadata),\n                'n_positive': metadata['label_positive'].sum() if 'label_positive' in metadata.columns else None,\n                'n_negative': (metadata['label_positive'] == 0).sum() if 'label_positive' in metadata.columns else None,\n                'balance_ratio': metadata['label_positive'].mean() if 'label_positive' in metadata.columns else None\n            }\n            summary_data.append(summary)\n    \n    summary_df = pd.DataFrame(summary_data)\n    \n    print(\"\\n📊 Dataset Übersicht:\")\n    print(summary_df)\n    \n    # Visualisierung\n    fig, axes = plt.subplots(1, 2, figsize=(15, 5))\n    \n    # Anzahl Repertoires pro Dataset\n    axes[0].bar(range(len(summary_df)), summary_df['n_repertoires'], color='skyblue')\n    axes[0].set_xlabel('Dataset Index')\n    axes[0].set_ylabel('Anzahl Repertoires')\n    axes[0].set_title('Repertoires pro Dataset')\n    \n    # Balance Ratio\n    if 'balance_ratio' in summary_df.columns:\n        axes[1].bar(range(len(summary_df)), summary_df['balance_ratio'], color='coral')\n        axes[1].axhline(y=0.5, color='red', linestyle='--', label='Perfect Balance')\n        axes[1].set_xlabel('Dataset Index')\n        axes[1].set_ylabel('Positive Label Ratio')\n        axes[1].set_title('Class Balance pro Dataset')\n        axes[1].legend()\n    \n    plt.tight_layout()\n    plt.show()\n    \n    return summary_df\n\n\n## 8. Speichere EDA-Ergebnisse\n\ndef save_eda_summary(summary_df, rep_stats, output_path='/kaggle/working/eda_summary.csv'):\n    \"\"\"Speichert EDA-Ergebnisse\"\"\"\n    summary_df.to_csv(output_path, index=False)\n    print(f\"\\n✅ EDA-Zusammenfassung gespeichert: {output_path}\")\n\n# 1. \ntrain_datasets, test_datasets = get_data_overview(train_datasets_dir, test_datasets_dir)\n\n# 2. Analysiere Training-Datensatz\nfor i in range(len(train_datasets)):\n    first_train_dataset = os.path.join(train_datasets_dir, train_datasets[i])\n    metadata_train, sample_repertoire = analyze_single_dataset(first_train_dataset, train_datasets[i])\n\n# 3. CDR3-Sequenz Analyse\n    sequences, lengths, v_genes, j_genes = analyze_cdr3_sequences(first_train_dataset, sample_size=10)\n\n## 4. Repertoire-Level Analyse\n    rep_stats = analyze_repertoire_characteristics(first_train_dataset, sample_size=30)\n\n# 5. Analysiere verschiedene K-mer Längen\n    for k in [3, 4, 5]:\n        kmer_counts = analyze_kmers(sequences, k=k, top_n=15)\n\n# 6. Vergleiche Datensatz\n    if train_datasets:\n        dataset_id = train_datasets[i].replace('train_dataset_', '')\n        compare_train_test_datasets(train_datasets_dir, test_datasets_dir, dataset_id)\n\n# 7. summary \nsummary = summarize_all_datasets(train_datasets_dir)\n\n# 8. save eda\nsave_eda_summary(summary, rep_stats)\n\nprint(\"\\n\" + \"=\" * 80)\nprint(\"✅ EDA ABGESCHLOSSEN\")\nprint(\"=\" * 80)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:02:48.097180Z","iopub.execute_input":"2025-11-30T06:02:48.097557Z","iopub.status.idle":"2025-11-30T06:04:17.250027Z","shell.execute_reply.started":"2025-11-30T06:02:48.097533Z","shell.execute_reply":"2025-11-30T06:04:17.248927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## 9. Sequenz-Diversität Analyse\nfrom scipy.stats import entropy\n\ndef calculate_diversity_metrics(dataset_path, sample_size=10):\n    \"\"\"Berechnet Diversitäts-Metriken\"\"\"\n    \n    \n    print(\"\\n\" + \"=\" * 80)\n    print(\"DIVERSITÄTS-ANALYSE\")\n    print(\"=\" * 80)\n    \n    metadata = pd.read_csv(os.path.join(dataset_path, 'metadata.csv'))\n    \n    diversity_stats = []\n    \n    for idx, row in metadata.head(sample_size).iterrows():\n        file_path = os.path.join(dataset_path, row['filename'])\n        df = pd.read_csv(file_path, sep='\\t')\n        \n        # Shannon Entropy\n        seq_counts = df['junction_aa'].value_counts()\n        shannon_ent = entropy(seq_counts, base=2)\n        \n        # Clonality (1 - normalized entropy)\n        max_entropy = np.log2(len(seq_counts))\n        clonality = 1 - (shannon_ent / max_entropy) if max_entropy > 0 else 0\n        \n        diversity_stats.append({\n            'repertoire_id': row['repertoire_id'],\n            'label': row.get('label_positive', None),\n            'unique_sequences': len(seq_counts),\n            'total_sequences': len(df),\n            'shannon_entropy': shannon_ent,\n            'clonality': clonality\n        })\n    \n    div_df = pd.DataFrame(diversity_stats)\n    \n    print(\"\\n📊 Diversitäts-Statistiken:\")\n    print(div_df.describe())\n    \n    # Visualisierung\n    if 'label' in div_df.columns:\n        fig, axes = plt.subplots(1, 2, figsize=(14, 5))\n        \n        div_df.boxplot(column='shannon_entropy', by='label', ax=axes[0])\n        axes[0].set_title('Shannon Entropy pro Label')\n        axes[0].set_xlabel('Label')\n        \n        div_df.boxplot(column='clonality', by='label', ax=axes[1])\n        axes[1].set_title('Clonality pro Label')\n        axes[1].set_xlabel('Label')\n        \n        plt.tight_layout()\n        plt.show()\n    \n    return div_df\n\ndiversity_df = calculate_diversity_metrics(first_train_dataset, sample_size=20)\n\n## 10. Motif-Suche\n\ndef find_common_motifs(sequences, motif_length=5, top_n=10):\n    \"\"\"Findet häufige Motive in Sequenzen\"\"\"\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(f\"MOTIF-ANALYSE (Länge: {motif_length})\")\n    print(\"=\" * 80)\n    \n    motif_positions = {}\n    \n    for seq in sequences[:5000]:\n        if isinstance(seq, str) and len(seq) >= motif_length:\n            for i in range(len(seq) - motif_length + 1):\n                motif = seq[i:i+motif_length]\n                position = 'N-terminal' if i < 3 else ('C-terminal' if i > len(seq) - motif_length - 3 else 'Middle')\n                \n                if motif not in motif_positions:\n                    motif_positions[motif] = {'count': 0, 'positions': []}\n                motif_positions[motif]['count'] += 1\n                motif_positions[motif]['positions'].append(position)\n    \n    # Sortiere nach Häufigkeit\n    sorted_motifs = sorted(motif_positions.items(), key=lambda x: x[1]['count'], reverse=True)[:top_n]\n    \n    print(f\"\\nTop {top_n} Motive:\")\n    for motif, data in sorted_motifs:\n        pos_dist = Counter(data['positions'])\n        print(f\"  {motif}: {data['count']}x - Positionen: {dict(pos_dist)}\")\n\nfind_common_motifs(sequences, motif_length=5, top_n=15)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:07:30.740006Z","iopub.execute_input":"2025-11-30T06:07:30.740351Z","iopub.status.idle":"2025-11-30T06:07:32.207160Z","shell.execute_reply.started":"2025-11-30T06:07:30.740326Z","shell.execute_reply":"2025-11-30T06:07:32.205816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## 11. Sequenz-Ähnlichkeitsanalyse (Clustering)\n\ndef analyze_sequence_similarity(sequences, sample_size=1000, method='levenshtein'):\n    \"\"\"\n    Analysiert Ähnlichkeiten zwischen Sequenzen\n    \"\"\"\n    from sklearn.cluster import DBSCAN\n    from sklearn.manifold import TSNE\n    from Levenshtein import distance as lev_distance\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(\"SEQUENZ-ÄHNLICHKEITSANALYSE\")\n    print(\"=\" * 80)\n    \n    # Sample für Performance\n    sample_seqs = [s for s in sequences if isinstance(s, str) and len(s) > 5][:sample_size]\n    \n    print(f\"\\n⏳ Berechne Distanzmatrix für {len(sample_seqs)} Sequenzen...\")\n    \n    # Erstelle Distanzmatrix\n    n = len(sample_seqs)\n    dist_matrix = np.zeros((n, n))\n    \n    for i in tqdm(range(n), desc=\"Berechne Distanzen\"):\n        for j in range(i+1, n):\n            dist = lev_distance(sample_seqs[i], sample_seqs[j])\n            # Normalisiere durch maximale Länge\n            max_len = max(len(sample_seqs[i]), len(sample_seqs[j]))\n            dist_matrix[i, j] = dist / max_len if max_len > 0 else 0\n            dist_matrix[j, i] = dist_matrix[i, j]\n    \n    print(f\"\\n📊 Distanz-Statistiken:\")\n    print(f\"   - Mittlere Distanz: {dist_matrix[dist_matrix > 0].mean():.3f}\")\n    print(f\"   - Std Distanz: {dist_matrix[dist_matrix > 0].std():.3f}\")\n    print(f\"   - Min Distanz: {dist_matrix[dist_matrix > 0].min():.3f}\")\n    print(f\"   - Max Distanz: {dist_matrix.max():.3f}\")\n    \n    # DBSCAN Clustering\n    print(\"\\n⏳ Führe DBSCAN Clustering durch...\")\n    clustering = DBSCAN(eps=0.3, min_samples=5, metric='precomputed')\n    clusters = clustering.fit_predict(dist_matrix)\n    \n    n_clusters = len(set(clusters)) - (1 if -1 in clusters else 0)\n    n_noise = list(clusters).count(-1)\n    \n    print(f\"\\n🔍 Clustering-Ergebnisse:\")\n    print(f\"   - Anzahl Cluster: {n_clusters}\")\n    print(f\"   - Noise-Punkte: {n_noise}\")\n    print(f\"   - Größter Cluster: {max(Counter(clusters).values())}\")\n    \n    # t-SNE Visualisierung\n    if len(sample_seqs) >= 50:\n        print(\"\\n⏳ Erstelle t-SNE Visualisierung...\")\n        tsne = TSNE(n_components=2, random_state=42, metric='precomputed')\n        coords = tsne.fit_transform(dist_matrix)\n        \n        plt.figure(figsize=(12, 8))\n        scatter = plt.scatter(coords[:, 0], coords[:, 1], \n                            c=clusters, cmap='viridis', \n                            alpha=0.6, s=50)\n        plt.colorbar(scatter, label='Cluster')\n        plt.title('t-SNE Visualisierung der Sequenz-Ähnlichkeit')\n        plt.xlabel('t-SNE Dimension 1')\n        plt.ylabel('t-SNE Dimension 2')\n        plt.tight_layout()\n        plt.show()\n    \n    return dist_matrix, clusters\n\n# Führe Ähnlichkeitsanalyse durch\n# Hinweis: Installiere python-Levenshtein falls noch nicht vorhanden\n# !pip install python-Levenshtein\n\ntry:\n    dist_matrix, clusters = analyze_sequence_similarity(sequences, sample_size=500)\nexcept ImportError:\n    print(\"⚠️  python-Levenshtein nicht installiert. Überspringe Ähnlichkeitsanalyse.\")\n    print(\"   Installiere mit: !pip install python-Levenshtein\")\n\n## 12. Positionsspezifische Aminosäure-Analyse\n\ndef analyze_positional_aa_distribution(sequences, max_length=20):\n    \"\"\"\n    Analysiert Aminosäure-Verteilung an jeder Position\n    \"\"\"\n    print(\"\\n\" + \"=\" * 80)\n    print(\"POSITIONSSPEZIFISCHE AMINOSÄURE-ANALYSE\")\n    print(\"=\" * 80)\n    \n    # Filter Sequenzen nach Länge\n    filtered_seqs = [s for s in sequences if isinstance(s, str) and len(s) == max_length]\n    \n    if len(filtered_seqs) < 100:\n        print(f\"⚠️  Zu wenige Sequenzen mit Länge {max_length}. Verwende variable Längen.\")\n        filtered_seqs = [s for s in sequences if isinstance(s, str) and 10 <= len(s) <= 25]\n        max_length = max([len(s) for s in filtered_seqs])\n    \n    print(f\"\\n📊 Analysiere {len(filtered_seqs)} Sequenzen mit Länge ~{max_length}\")\n    \n    # Erstelle Positions-Matrix\n    aa_list = list('ACDEFGHIKLMNPQRSTVWY')\n    position_matrix = np.zeros((max_length, len(aa_list)))\n    \n    for seq in filtered_seqs[:5000]:  # Sample\n        for pos, aa in enumerate(seq[:max_length]):\n            if aa in aa_list:\n                aa_idx = aa_list.index(aa)\n                position_matrix[pos, aa_idx] += 1\n    \n    # Normalisiere\n    position_matrix = position_matrix / position_matrix.sum(axis=1, keepdims=True)\n    \n    # Heatmap\n    plt.figure(figsize=(16, 8))\n    sns.heatmap(position_matrix.T, \n                xticklabels=range(1, max_length+1),\n                yticklabels=aa_list,\n                cmap='YlOrRd',\n                cbar_kws={'label': 'Häufigkeit'})\n    plt.xlabel('Position in der Sequenz')\n    plt.ylabel('Aminosäure')\n    plt.title('Positionsspezifische Aminosäure-Verteilung')\n    plt.tight_layout()\n    plt.show()\n    \n    # Finde konservierte Positionen\n    print(\"\\n🔍 Konservierte Positionen (Entropie < 2.0):\")\n    for pos in range(max_length):\n        pos_entropy = entropy(position_matrix[pos] + 1e-10, base=2)\n        if pos_entropy < 2.0:\n            top_aa = aa_list[np.argmax(position_matrix[pos])]\n            top_freq = position_matrix[pos].max()\n            print(f\"   Position {pos+1}: {top_aa} ({top_freq*100:.1f}%), Entropie: {pos_entropy:.2f}\")\n    \n    return position_matrix\n\npos_matrix = analyze_positional_aa_distribution(sequences, max_length=15)\n\n## 13. V-J Gene Kombinationsanalyse\n\ndef analyze_vj_combinations(dataset_path, min_count=10):\n    \"\"\"\n    Analysiert V-J Gen-Kombinationen und deren Assoziation mit Labels\n    \"\"\"\n    print(\"\\n\" + \"=\" * 80)\n    print(\"V-J GEN-KOMBINATIONSANALYSE\")\n    print(\"=\" * 80)\n    \n    metadata = pd.read_csv(os.path.join(dataset_path, 'metadata.csv'))\n    \n    vj_combinations_pos = Counter()\n    vj_combinations_neg = Counter()\n    \n    print(f\"\\n⏳ Analysiere V-J Kombinationen...\")\n    \n    for idx, row in tqdm(metadata.iterrows(), total=len(metadata)):\n        file_path = os.path.join(dataset_path, row['filename'])\n        df = pd.read_csv(file_path, sep='\\t')\n        \n        if 'v_call' in df.columns and 'j_call' in df.columns:\n            for _, seq_row in df.iterrows():\n                v_gene = seq_row['v_call']\n                j_gene = seq_row['j_call']\n                \n                if pd.notna(v_gene) and pd.notna(j_gene):\n                    combination = f\"{v_gene}_{j_gene}\"\n                    \n                    if row.get('label_positive', None) == 1:\n                        vj_combinations_pos[combination] += 1\n                    else:\n                        vj_combinations_neg[combination] += 1\n    \n    # Berechne Enrichment\n    all_combinations = set(vj_combinations_pos.keys()) | set(vj_combinations_neg.keys())\n    \n    enrichment_data = []\n    for combo in all_combinations:\n        pos_count = vj_combinations_pos[combo]\n        neg_count = vj_combinations_neg[combo]\n        total = pos_count + neg_count\n        \n        if total >= min_count:\n            enrichment = (pos_count / (pos_count + neg_count)) if total > 0 else 0\n            enrichment_data.append({\n                'combination': combo,\n                'positive_count': pos_count,\n                'negative_count': neg_count,\n                'total': total,\n                'enrichment_score': enrichment\n            })\n    \n    enrichment_df = pd.DataFrame(enrichment_data).sort_values('enrichment_score', ascending=False)\n    \n    print(f\"\\n📊 Gefundene V-J Kombinationen: {len(enrichment_df)}\")\n    print(f\"\\nTop 10 in Positiven angereichert:\")\n    print(enrichment_df.head(10))\n    print(f\"\\nTop 10 in Negativen angereichert:\")\n    print(enrichment_df.tail(10))\n    \n    # Visualisierung\n    fig, axes = plt.subplots(1, 2, figsize=(16, 6))\n    \n    # Top enriched in positive\n    top_pos = enrichment_df.head(15)\n    axes[0].barh(range(len(top_pos)), top_pos['enrichment_score'], color='green', alpha=0.7)\n    axes[0].set_yticks(range(len(top_pos)))\n    axes[0].set_yticklabels(top_pos['combination'], fontsize=8)\n    axes[0].set_xlabel('Enrichment Score (Positive)')\n    axes[0].set_title('Top 15 V-J Kombinationen in Positiven')\n    axes[0].invert_yaxis()\n    \n    # Top enriched in negative\n    top_neg = enrichment_df.tail(15)\n    axes[1].barh(range(len(top_neg)), top_neg['enrichment_score'], color='red', alpha=0.7)\n    axes[1].set_yticks(range(len(top_neg)))\n    axes[1].set_yticklabels(top_neg['combination'], fontsize=8)\n    axes[1].set_xlabel('Enrichment Score (Negative)')\n    axes[1].set_title('Top 15 V-J Kombinationen in Negativen')\n    axes[1].invert_yaxis()\n    \n    plt.tight_layout()\n    plt.show()\n    \n    return enrichment_df\n\nvj_enrichment = analyze_vj_combinations(first_train_dataset, min_count=5)\n\n## 14. Sequenz-Logo (Motif Visualisierung)\n\ndef create_sequence_logo(sequences, position_range=(0, 15)):\n    \"\"\"\n    Erstellt ein Sequenz-Logo für konservierte Regionen\n    \"\"\"\n    print(\"\\n\" + \"=\" * 80)\n    print(\"SEQUENZ-LOGO ANALYSE\")\n    print(\"=\" * 80)\n    \n    # Filtere Sequenzen\n    start, end = position_range\n    filtered_seqs = [s[start:end] for s in sequences \n                     if isinstance(s, str) and len(s) >= end][:1000]\n    \n    if not filtered_seqs:\n        print(\"⚠️  Keine passenden Sequenzen gefunden\")\n        return\n    \n    # Berechne Positional Frequency Matrix\n    aa_list = list('ACDEFGHIKLMNPQRSTVWY')\n    length = end - start\n    pfm = np.zeros((length, len(aa_list)))\n    \n    for seq in filtered_seqs:\n        for pos, aa in enumerate(seq[:length]):\n            if aa in aa_list:\n                pfm[pos, aa_list.index(aa)] += 1\n    \n    # Konvertiere zu PWM (Position Weight Matrix)\n    pfm = pfm + 0.01  # Pseudocount\n    pwm = pfm / pfm.sum(axis=1, keepdims=True)\n    \n    # Berechne Information Content\n    max_entropy = np.log2(len(aa_list))\n    ic = np.zeros(length)\n    \n    for pos in range(length):\n        pos_entropy = entropy(pwm[pos], base=2)\n        ic[pos] = max_entropy - pos_entropy\n    \n    # Visualisierung\n    fig, axes = plt.subplots(2, 1, figsize=(15, 8))\n    \n    # Information Content\n    axes[0].bar(range(length), ic, color='steelblue', alpha=0.7)\n    axes[0].set_xlabel('Position')\n    axes[0].set_ylabel('Information Content (bits)')\n    axes[0].set_title('Sequenz-Konservierung pro Position')\n    axes[0].axhline(y=1.0, color='red', linestyle='--', alpha=0.5, label='Threshold')\n    axes[0].legend()\n    \n    # Heatmap der Aminosäure-Häufigkeiten\n    sns.heatmap(pwm.T, ax=axes[1], \n                xticklabels=range(start+1, end+1),\n                yticklabels=aa_list,\n                cmap='Blues',\n                cbar_kws={'label': 'Wahrscheinlichkeit'})\n    axes[1].set_xlabel('Position')\n    axes[1].set_ylabel('Aminosäure')\n    axes[1].set_title('Aminosäure-Wahrscheinlichkeiten pro Position')\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Konsensus-Sequenz\n    consensus = ''.join([aa_list[np.argmax(pwm[pos])] for pos in range(length)])\n    print(f\"\\n🧬 Konsensus-Sequenz: {consensus}\")\n    print(f\"   Mittlerer Information Content: {ic.mean():.2f} bits\")\n    \n    return pwm, ic, consensus\n\npwm, ic, consensus = create_sequence_logo(sequences, position_range=(0, 15))\n\n## 15. Label-spezifische Sequenzanalyse\n\ndef compare_sequences_by_label(dataset_path, sample_size=1000):\n    \"\"\"\n    Vergleicht Sequenz-Charakteristiken zwischen positiven und negativen Labels\n    \"\"\"\n    print(\"\\n\" + \"=\" * 80)\n    print(\"LABEL-SPEZIFISCHE SEQUENZANALYSE\")\n    print(\"=\" * 80)\n    \n    metadata = pd.read_csv(os.path.join(dataset_path, 'metadata.csv'))\n    \n    pos_sequences = []\n    neg_sequences = []\n    \n    print(f\"\\n⏳ Sammle Sequenzen nach Label...\")\n    \n    for idx, row in tqdm(metadata.iterrows(), total=len(metadata)):\n        file_path = os.path.join(dataset_path, row['filename'])\n        df = pd.read_csv(file_path, sep='\\t')\n        \n        sequences = df['junction_aa'].dropna().tolist()\n        \n        if row.get('label_positive', None) == 1:\n            pos_sequences.extend(sequences[:sample_size // len(metadata)])\n        else:\n            neg_sequences.extend(sequences[:sample_size // len(metadata)])\n    \n    print(f\"\\n📊 Gesammelte Sequenzen:\")\n    print(f\"   - Positive: {len(pos_sequences)}\")\n    print(f\"   - Negative: {len(neg_sequences)}\")\n    \n    # Vergleiche Längenverteilungen\n    pos_lengths = [len(s) for s in pos_sequences if isinstance(s, str)]\n    neg_lengths = [len(s) for s in neg_sequences if isinstance(s, str)]\n    \n    # Vergleiche Aminosäure-Kompositionen\n    pos_aa_counter = Counter()\n    neg_aa_counter = Counter()\n    \n    for seq in pos_sequences[:5000]:\n        if isinstance(seq, str):\n            pos_aa_counter.update(seq)\n    \n    for seq in neg_sequences[:5000]:\n        if isinstance(seq, str):\n            neg_aa_counter.update(seq)\n    \n    # Normalisiere\n    pos_total = sum(pos_aa_counter.values())\n    neg_total = sum(neg_aa_counter.values())\n    \n    aa_comparison = []\n    for aa in 'ACDEFGHIKLMNPQRSTVWY':\n        pos_freq = pos_aa_counter[aa] / pos_total if pos_total > 0 else 0\n        neg_freq = neg_aa_counter[aa] / neg_total if neg_total > 0 else 0\n        fold_change = np.log2((pos_freq + 1e-10) / (neg_freq + 1e-10))\n        \n        aa_comparison.append({\n            'aa': aa,\n            'pos_freq': pos_freq,\n            'neg_freq': neg_freq,\n            'fold_change': fold_change\n        })\n    \n    aa_comp_df = pd.DataFrame(aa_comparison).sort_values('fold_change', ascending=False)\n    \n    # Visualisierung\n    fig, axes = plt.subplots(2, 2, figsize=(16, 10))\n    \n    # 1. Längenverteilung\n    axes[0, 0].hist(pos_lengths, bins=30, alpha=0.6, label='Positive', color='green', density=True)\n    axes[0, 0].hist(neg_lengths, bins=30, alpha=0.6, label='Negative', color='red', density=True)\n    axes[0, 0].set_xlabel('Sequenzlänge')\n    axes[0, 0].set_ylabel('Dichte')\n    axes[0, 0].set_title('Längenverteilung nach Label')\n    axes[0, 0].legend()\n    \n    # 2. Boxplot Längen\n    axes[0, 1].boxplot([pos_lengths, neg_lengths], labels=['Positive', 'Negative'])\n    axes[0, 1].set_ylabel('Sequenzlänge')\n    axes[0, 1].set_title('Längen-Vergleich')\n    \n    # 3. Aminosäure Fold-Change\n    colors = ['green' if fc > 0 else 'red' for fc in aa_comp_df['fold_change']]\n    axes[1, 0].bar(aa_comp_df['aa'], aa_comp_df['fold_change'], color=colors, alpha=0.7)\n    axes[1, 0].axhline(y=0, color='black', linestyle='-', linewidth=0.5)\n    axes[1, 0].set_xlabel('Aminosäure')\n    axes[1, 0].set_ylabel('Log2 Fold Change (Pos/Neg)')\n    axes[1, 0].set_title('Aminosäure-Anreicherung')\n    \n    # 4. Frequenz-Vergleich\n    x = np.arange(len(aa_comp_df))\n    width = 0.35\n    axes[1, 1].bar(x - width/2, aa_comp_df['pos_freq'], width, label='Positive', color='green', alpha=0.7)\n    axes[1, 1].bar(x + width/2, aa_comp_df['neg_freq'], width, label='Negative', color='red', alpha=0.7)\n    axes[1, 1].set_xlabel('Aminosäure')\n    axes[1, 1].set_ylabel('Frequenz')\n    axes[1, 1].set_title('Aminosäure-Frequenzen')\n    axes[1, 1].set_xticks(x)\n    axes[1, 1].set_xticklabels(aa_comp_df['aa'])\n    axes[1, 1].legend()\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Statistischer Test\n    from scipy.stats import mannwhitneyu\n    \n    stat, pval = mannwhitneyu(pos_lengths, neg_lengths)\n    print(f\"\\n📈 Mann-Whitney U Test (Längen):\")\n    print(f\"   - Statistik: {stat:.2f}\")\n    print(f\"   - P-Wert: {pval:.4e}\")\n    print(f\"   - Signifikant: {'Ja' if pval < 0.05 else 'Nein'}\")\n    \n    print(f\"\\n🧬 Top 5 in Positiven angereicherte Aminosäuren:\")\n    print(aa_comp_df.head(5)[['aa', 'fold_change']])\n    \n    print(f\"\\n🧬 Top 5 in Negativen angereicherte Aminosäuren:\")\n    print(aa_comp_df.tail(5)[['aa', 'fold_change']])\n    \n    return aa_comp_df, pos_sequences, neg_sequences\n\naa_comparison, pos_seqs, neg_seqs = compare_sequences_by_label(first_train_dataset, sample_size=2000)\n\n## 16. K-mer Enrichment Analyse\n\ndef kmer_enrichment_analysis(pos_sequences, neg_sequences, k=4, top_n=30):\n    \"\"\"\n    Findet K-mere, die in einer Klasse angereichert sind\n    \"\"\"\n    print(\"\\n\" + \"=\" * 80)\n    print(f\"{k}-MER ENRICHMENT ANALYSE\")\n    print(\"=\" * 80)\n    \n    pos_kmers = Counter()\n    neg_kmers = Counter()\n    \n    print(f\"\\n⏳ Extrahiere {k}-mere...\")\n    \n    # Positive Sequenzen\n    for seq in pos_sequences[:5000]:\n        if isinstance(seq, str):\n            for i in range(len(seq) - k + 1):\n                pos_kmers[seq[i:i+k]] += 1\n    \n    # Negative Sequenzen\n    for seq in neg_sequences[:5000]:\n        if isinstance(seq, str):\n            for i in range(len(seq) - k + 1):\n                neg_kmers[seq[i:i+k]] += 1\n    \n    # Berechne Enrichment\n    all_kmers = set(pos_kmers.keys()) | set(neg_kmers.keys())\n    \n    enrichment_data = []\n    for kmer in all_kmers:\n        pos_count = pos_kmers[kmer]\n        neg_count = neg_kmers[kmer]\n        total = pos_count + neg_count\n        \n        if total >= 10:  # Minimum count threshold\n            # Berechne Odds Ratio\n            pos_total = sum(pos_kmers.values())\n            neg_total = sum(neg_kmers.values())\n            \n            pos_freq = pos_count / pos_total\n            neg_freq = neg_count / neg_total\n            \n            if neg_freq > 0:\n                odds_ratio = pos_freq / neg_freq\n                log_odds = np.log2(odds_ratio)\n            else:\n                log_odds = 10  # Sehr hoher Wert wenn nur in pos\n            \n            enrichment_data.append({\n                'kmer': kmer,\n                'pos_count': pos_count,\n                'neg_count': neg_count,\n                'log2_odds_ratio': log_odds,\n                'pos_freq': pos_freq,\n                'neg_freq': neg_freq\n            })\n    \n    enrich_df = pd.DataFrame(enrichment_data).sort_values('log2_odds_ratio', ascending=False)\n    \n    print(f\"\\n📊 Analysierte {k}-mere: {len(enrich_df)}\")\n    \n    print(f\"\\nTop {top_n//2} in Positiven angereichert:\")\n    print(enrich_df.head(top_n//2)[['kmer', 'log2_odds_ratio', 'pos_count', 'neg_count']])\n    \n    print(f\"\\nTop {top_n//2} in Negativen angereichert:\")\n    print(enrich_df.tail(top_n//2)[['kmer', 'log2_odds_ratio', 'pos_count', 'neg_count']])\n    \n    # Visualisierung\n    fig, axes = plt.subplots(1, 2, figsize=(16, 6))\n    \n    # Volcano Plot\n    axes[0].scatter(enrich_df['log2_odds_ratio'], \n                   -np.log10(1/(enrich_df['pos_count'] + enrich_df['neg_count'])),\n                   alpha=0.5, s=20)\n    axes[0].axvline(x=0, color='red', linestyle='--', alpha=0.5)\n    axes[0].set_xlabel('Log2 Odds Ratio')\n    axes[0].set_ylabel('-Log10(1/Total Count)')\n    axes[0].set_title(f'{k}-mer Enrichment Volcano Plot')\n    \n    # Top enriched\n    top_pos = enrich_df.head(15)\n    top_neg = enrich_df.tail(15)\n    combined = pd.concat([top_pos, top_neg])\n    \n    colors = ['green' if x > 0 else 'red' for x in combined['log2_odds_ratio']]\n    axes[1].barh(range(len(combined)), combined['log2_odds_ratio'], color=colors, alpha=0.7)\n    axes[1].set_yticks(range(len(combined)))\n    axes[1].set_yticklabels(combined['kmer'], fontsize=8)\n    axes[1].axvline(x=0, color='black', linestyle='-', linewidth=0.5)\n    axes[1].set_xlabel('Log2 Odds Ratio')\n    axes[1].set_title(f'Top {k}-mere nach Enrichment')\n    axes[1].invert_yaxis()\n    \n    plt.tight_layout()\n    plt.show()\n    \n    return enrich_df\n\n# Führe für verschiedene k-Werte durch\nfor k_val in [3, 4, 5]:\n    kmer_enrich = kmer_enrichment_analysis(pos_seqs, neg_seqs, k=k_val, top_n=20)\n\n## 17. Hydrophobizitäts- und Ladungsanalyse\n\ndef analyze_physicochemical_properties(sequences, labels=None):\n    \"\"\"\n    Analysiert physikochemische Eigenschaften der Sequenzen\n    \"\"\"\n    print(\"\\n\" + \"=\" * 80)\n    print(\"PHYSIKOCHEMISCHE EIGENSCHAFTEN\")\n    print(\"=\" * 80)\n    \n    # Aminosäure-Eigenschaften\n    hydrophobicity = {\n        'A': 1.8, 'C': 2.5, 'D': -3.5, 'E': -3.5, 'F': 2.8,\n        'G': -0.4, 'H': -3.2, 'I': 4.5, 'K': -3.9, 'L': 3.8,\n        'M': 1.9, 'N': -3.5, 'P': -1.6, 'Q': -3.5, 'R': -4.5,\n        'S': -0.8, 'T': -0.7, 'V': 4.2, 'W': -0.9, 'Y': -1.3\n    }\n    \n    charge = {\n        'A': 0, 'C': 0, 'D': -1, 'E': -1, 'F': 0,\n        'G': 0, 'H': 0.5, 'I': 0, 'K': 1, 'L': 0,\n        'M': 0, 'N': 0, 'P': 0, 'Q': 0, 'R': 1,\n        'S': 0, 'T': 0, 'V': 0, 'W': 0, 'Y': 0\n    }\n    \n    properties = []\n    \n    print(f\"\\n⏳ Berechne Eigenschaften für {len(sequences)} Sequenzen...\")\n    \n    for seq in sequences[:5000]:\n        if isinstance(seq, str) and len(seq) > 0:\n            # Hydrophobizität\n            hydro = np.mean([hydrophobicity.get(aa, 0) for aa in seq])\n            \n            # Ladung\n            net_charge = sum([charge.get(aa, 0) for aa in seq])\n            \n            # Aromatizität (F, W, Y)\n            aromatic = sum([1 for aa in seq if aa in 'FWY']) / len(seq)\n            \n            # Polarität\n            polar = sum([1 for aa in seq if aa in 'STNQ']) / len(seq)\n            \n            # Basisch\n            basic = sum([1 for aa in seq if aa in 'KRH']) / len(seq)\n            \n            # Sauer\n            acidic = sum([1 for aa in seq if aa in 'DE']) / len(seq)\n            \n            properties.append({\n                'hydrophobicity': hydro,\n                'net_charge': net_charge,\n                'aromatic_content': aromatic,\n                'polar_content': polar,\n                'basic_content': basic,\n                'acidic_content': acidic,\n                'length': len(seq)\n            })\n    \n    prop_df = pd.DataFrame(properties)\n    \n    print(f\"\\n📊 Physikochemische Statistiken:\")\n    print(prop_df.describe())\n    \n    # Visualisierung\n    fig, axes = plt.subplots(2, 3, figsize=(18, 10))\n    axes = axes.flatten()\n    \n    properties_to_plot = ['hydrophobicity', 'net_charge', 'aromatic_content', \n                          'polar_content', 'basic_content', 'acidic_content']\n    \n    for idx, prop in enumerate(properties_to_plot):\n        axes[idx].hist(prop_df[prop], bins=50, color='skyblue', edgecolor='black', alpha=0.7)\n        axes[idx].axvline(prop_df[prop].mean(), color='red', linestyle='--', \n                         label=f'Mean: {prop_df[prop].mean():.3f}')\n        axes[idx].set_xlabel(prop.replace('_', ' ').title())\n        axes[idx].set_ylabel('Häufigkeit')\n        axes[idx].set_title(f'Verteilung: {prop.replace(\"_\", \" \").title()}')\n        axes[idx].legend()\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Korrelationsanalyse\n    corr_matrix = prop_df.corr()\n    \n    plt.figure(figsize=(10, 8))\n    sns.heatmap(corr_matrix, annot=True, cmap='coolwarm', center=0, \n                square=True, linewidths=1)\n    plt.title('Korrelation zwischen physikochemischen Eigenschaften')\n    plt.tight_layout()\n    plt.show()\n    \n    return prop_df\n\nprop_df = analyze_physicochemical_properties(sequences)\n\n## 18. Speichere umfassende EDA-Ergebnisse\n\ndef save_comprehensive_eda_results(output_dir='/kaggle/working/eda_results'):\n    \"\"\"\n    Speichert alle EDA-Ergebnisse\n    \"\"\"\n    os.makedirs(output_dir, exist_ok=True)\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(\"SPEICHERE EDA-ERGEBNISSE\")\n    print(\"=\" * 80)\n    \n    # Speichere DataFrames\n    results_to_save = {\n        'dataset_summary.csv': summary,\n        'repertoire_stats.csv': rep_stats,\n        'vj_enrichment.csv': vj_enrichment,\n        'aa_comparison.csv': aa_comparison,\n        'kmer_enrichment.csv': kmer_enrich,\n        'physicochemical_properties.csv': prop_df,\n        'diversity_metrics.csv': diversity_df\n    }\n    \n    for filename, df in results_to_save.items():\n        filepath = os.path.join(output_dir, filename)\n        df.to_csv(filepath, index=False)\n        print(f\"✅ Gespeichert: {filepath}\")\n    \n    print(f\"\\n✅ Alle EDA-Ergebnisse gespeichert in: {output_dir}\")\n\nsave_comprehensive_eda_results()\n\nprint(\"\\n\" + \"=\" * 80)\nprint(\"🎉 VOLLSTÄNDIGE EDA ABGESCHLOSSEN!\")\nprint(\"=\" * 80)\nprint(\"\\n📋 Zusammenfassung der durchgeführten Analysen:\")\nprint(\"   1. ✅ Datenstruktur-Überblick\")\nprint(\"   2. ✅ Einzeldatensatz-Analyse\")\nprint(\"   3. ✅ CDR3-Sequenz-Analyse\")\nprint(\"   4. ✅ Repertoire-Level-Analyse\")\nprint(\"   5. ✅ K-mer-Analyse\")\nprint(\"   6. ✅ Train/Test-Vergleich\")\nprint(\"   7. ✅ Dataset-Zusammenfassung\")\nprint(\"   8. ✅ Diversitäts-Metriken\")\nprint(\"   9. ✅ Motif-Suche\")\nprint(\"   10. ✅ Sequenz-Ähnlichkeit\")\nprint(\"   11. ✅ Positionsspezifische Analyse\")\nprint(\"   12. ✅ V-J Gen-Kombinationen\")\nprint(\"   13. ✅ Sequenz-Logo\")\nprint(\"   14. ✅ Label-spezifische Analyse\")\nprint(\"   15. ✅ K-mer Enrichment\")\nprint(\"   16. ✅ Physikochemische Eigenschaften\")\nprint(\"   17. ✅ Ergebnisse gespeichert\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:07:32.208933Z","iopub.execute_input":"2025-11-30T06:07:32.209947Z","iopub.status.idle":"2025-11-30T06:17:52.381416Z","shell.execute_reply.started":"2025-11-30T06:07:32.209920Z","shell.execute_reply":"2025-11-30T06:17:52.379979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## 20. Erstelle einen HTML-Report\n\ndef create_eda_html_report(output_path='/kaggle/working/eda_report.html'):\n    \"\"\"\n    Erstellt einen HTML-Report mit allen wichtigen Erkenntnissen\n    \"\"\"\n    \n    html_content = f\"\"\"\n    <!DOCTYPE html>\n    <html>\n    <head>\n        <title>EDA Report - Adaptive Immune Profiling Challenge</title>\n        <style>\n            body {{ font-family: Arial, sans-serif; margin: 40px; background-color: #f5f5f5; }}\n            h1 {{ color: #2c3e50; border-bottom: 3px solid #3498db; padding-bottom: 10px; }}\n            h2 {{ color: #34495e; margin-top: 30px; }}\n            .metric {{ background-color: white; padding: 20px; margin: 10px 0; border-radius: 5px; box-shadow: 0 2px 4px rgba(0,0,0,0.1); }}\n            .metric-value {{ font-size: 24px; font-weight: bold; color: #3498db; }}\n            .metric-label {{ color: #7f8c8d; font-size: 14px; }}\n            table {{ border-collapse: collapse; width: 100%; margin: 20px 0; background-color: white; }}\n            th, td {{ border: 1px solid #ddd; padding: 12px; text-align: left; }}\n            th {{ background-color: #3498db; color: white; }}\n            .highlight {{ background-color: #fff3cd; padding: 15px; border-left: 4px solid #ffc107; margin: 20px 0; }}\n        </style>\n    </head>\n    <body>\n        <h1>🧬 Explorative Datenanalyse - Adaptive Immune Profiling Challenge 2025</h1>\n        \n        <div class=\"metric\">\n            <div class=\"metric-label\">Anzahl Training Datasets</div>\n            <div class=\"metric-value\">{len(train_datasets)}</div>\n        </div>\n        \n        <div class=\"metric\">\n            <div class=\"metric-label\">Anzahl Test Datasets</div>\n            <div class=\"metric-value\">{len(test_datasets)}</div>\n        </div>\n        \n        <div class=\"metric\">\n            <div class=\"metric-label\">Durchschnittliche Sequenzlänge</div>\n            <div class=\"metric-value\">{np.mean(lengths):.1f} ± {np.std(lengths):.1f}</div>\n        </div>\n        \n        <h2>📊 Wichtigste Erkenntnisse</h2>\n        \n        <div class=\"highlight\">\n            <strong>1. Sequenzlängen:</strong> Die meisten CDR3-Sequenzen haben eine Länge zwischen \n            {int(np.percentile(lengths, 25))} und {int(np.percentile(lengths, 75))} Aminosäuren.\n        </div>\n        \n        <div class=\"highlight\">\n            <strong>2. Aminosäure-Komposition:</strong> Die häufigsten Aminosäuren sind \n            {', '.join([aa for aa, _ in Counter(''.join(str(s) for s in sequences[:1000])).most_common(5)])}.\n        </div>\n        \n        <div class=\"highlight\">\n            <strong>3. Class Balance:</strong> Die Datasets zeigen unterschiedliche Balance-Verhältnisse \n            zwischen positiven und negativen Labels.\n        </div>\n        \n        <h2>📈 Empfehlungen für Modellierung</h2>\n        <ul>\n            <li>✅ Verwenden Sie K-mer Features (k=3,4,5) als Basis-Features</li>\n            <li>✅ Berücksichtigen Sie V-J Gen-Kombinationen</li>\n            <li>✅ Nutzen Sie Repertoire-Level Aggregationen</li>\n            <li>✅ Implementieren Sie Class-Balancing Strategien</li>\n            <li>✅ Extrahieren Sie physikochemische Eigenschaften</li>\n        </ul>\n        \n        <h2>📁 Generierte Dateien</h2>\n        <ul>\n            <li>dataset_summary.csv</li>\n            <li>repertoire_stats.csv</li>\n            <li>vj_enrichment.csv</li>\n            <li>aa_comparison.csv</li>\n            <li>kmer_enrichment.csv</li>\n            <li>physicochemical_properties.csv</li>\n        </ul>\n        \n        <p style=\"margin-top: 50px; color: #7f8c8d; font-size: 12px;\">\n            Generiert am: {pd.Timestamp.now().strftime('%Y-%m-%d %H:%M:%S')}\n        </p>\n    </body>\n    </html>\n    \"\"\"\n    \n    with open(output_path, 'w', encoding='utf-8') as f:\n        f.write(html_content)\n    \n    print(f\"\\n✅ HTML-Report erstellt: {output_path}\")\n\ncreate_eda_html_report()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T06:17:52.383322Z","iopub.execute_input":"2025-11-30T06:17:52.383630Z","iopub.status.idle":"2025-11-30T06:17:52.472524Z","shell.execute_reply.started":"2025-11-30T06:17:52.383607Z","shell.execute_reply":"2025-11-30T06:17:52.471381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Optimized Feature Engineering with Dimensionality Reduction\n\nclass OptimizedFeatureEngineering(FeatureEngineering):\n    \"\"\"\n    Optimized feature engineering with dimensionality reduction\n    \"\"\"\n    \n    def __init__(self, kmer_sizes: List[int] = [3, 4], \n                 use_vj_features: bool = True,\n                 use_physicochemical: bool = True,\n                 use_diversity: bool = True,\n                 max_kmer_features: int = 1000,\n                 use_pca: bool = False,\n                 pca_components: int = 100):\n        \"\"\"\n        Initialize optimized feature engineering\n        \n        Args:\n            kmer_sizes: List of k-mer sizes (reduced to [3,4] for speed)\n            use_vj_features: Whether to include V-J gene features\n            use_physicochemical: Whether to include physicochemical properties\n            use_diversity: Whether to include diversity metrics\n            max_kmer_features: Maximum number of k-mer features to keep\n            use_pca: Whether to use PCA for dimensionality reduction\n            pca_components: Number of PCA components\n        \"\"\"\n        super().__init__(kmer_sizes, use_vj_features, use_physicochemical, use_diversity)\n        self.max_kmer_features = max_kmer_features\n        self.use_pca = use_pca\n        self.pca_components = pca_components\n        self.pca = None\n        self.top_kmers = {}\n        self.top_vj = {}\n    \n    def select_top_features(self, features_df: pd.DataFrame, labels: pd.Series) -> pd.DataFrame:\n        \"\"\"\n        Select top features based on variance and correlation with labels\n        \n        Args:\n            features_df: Features DataFrame\n            labels: Target labels\n            \n        Returns:\n            Reduced features DataFrame\n        \"\"\"\n        print(\"Selecting top features...\")\n        \n        # Separate feature types\n        kmer_cols = [col for col in features_df.columns if col.startswith('kmer_')]\n        vj_cols = [col for col in features_df.columns if col.startswith('vj_')]\n        other_cols = [col for col in features_df.columns if not (col.startswith('kmer_') or col.startswith('vj_'))]\n        \n        selected_cols = other_cols.copy()  # Always keep physicochemical and diversity features\n        \n        # Select top k-mer features based on variance\n        if kmer_cols:\n            kmer_df = features_df[kmer_cols]\n            variances = kmer_df.var()\n            top_kmer_cols = variances.nlargest(min(self.max_kmer_features, len(kmer_cols))).index.tolist()\n            selected_cols.extend(top_kmer_cols)\n            \n            # Store for transform\n            self.top_kmers = {col: True for col in top_kmer_cols}\n        \n        # Select top V-J features based on variance\n        if vj_cols:\n            vj_df = features_df[vj_cols]\n            variances = vj_df.var()\n            top_vj_cols = variances.nlargest(min(500, len(vj_cols))).index.tolist()\n            selected_cols.extend(top_vj_cols)\n            \n            # Store for transform\n            self.top_vj = {col: True for col in top_vj_cols}\n        \n        reduced_df = features_df[selected_cols]\n        \n        print(f\"Reduced features from {len(features_df.columns)} to {len(selected_cols)}\")\n        \n        return reduced_df\n    \n    def fit_transform(self, data_dir: str) -> Tuple[pd.DataFrame, pd.Series]:\n        \"\"\"\n        Fit feature engineering pipeline and transform training data\n        \n        Args:\n            data_dir: Path to training data directory\n            \n        Returns:\n            Tuple of (features_df, labels)\n        \"\"\"\n        print(\"Extracting features from training data (optimized)...\")\n        \n        metadata = pd.read_csv(os.path.join(data_dir, 'metadata.csv'))\n        \n        all_features = []\n        all_labels = []\n        repertoire_ids = []\n        \n        # Use parallel processing for feature extraction\n        from joblib import Parallel, delayed\n        \n        def process_repertoire(row):\n            \"\"\"Process single repertoire\"\"\"\n            file_path = os.path.join(data_dir, row['filename'])\n            repertoire_df = pd.read_csv(file_path, sep='\\t')\n            features = self.extract_repertoire_features(repertoire_df)\n            return features, row['label_positive'], row['repertoire_id']\n        \n        # Parallel processing\n        results = Parallel(n_jobs=self.n_jobs if hasattr(self, 'n_jobs') else 4)(\n            delayed(process_repertoire)(row) \n            for _, row in tqdm(metadata.iterrows(), total=len(metadata), desc=\"Processing repertoires\")\n        )\n        \n        # Unpack results\n        for features, label, rep_id in results:\n            all_features.append(features)\n            all_labels.append(label)\n            repertoire_ids.append(rep_id)\n        \n        # Convert to DataFrame\n        features_df = pd.DataFrame(all_features)\n        features_df.index = repertoire_ids\n        features_df = features_df.fillna(0)\n        \n        labels = pd.Series(all_labels, index=repertoire_ids)\n        \n        # Feature selection\n        features_df = self.select_top_features(features_df, labels)\n        \n        # Store feature names\n        self.feature_names = features_df.columns.tolist()\n        \n        # Apply PCA if requested\n        if self.use_pca:\n            print(f\"Applying PCA (n_components={self.pca_components})...\")\n            self.pca = PCA(n_components=min(self.pca_components, len(features_df.columns)))\n            features_pca = self.pca.fit_transform(features_df)\n            \n            # Create new DataFrame with PCA components\n            pca_cols = [f'pca_{i}' for i in range(features_pca.shape[1])]\n            features_df = pd.DataFrame(features_pca, columns=pca_cols, index=features_df.index)\n            self.feature_names = pca_cols\n            \n            print(f\"Explained variance ratio: {self.pca.explained_variance_ratio_.sum():.3f}\")\n        else:\n            # Fit scaler\n            self.scaler = StandardScaler()\n            features_scaled = self.scaler.fit_transform(features_df)\n            features_df = pd.DataFrame(features_scaled, columns=features_df.columns, index=features_df.index)\n        \n        print(f\"Final feature count: {len(self.feature_names)}\")\n        \n        return features_df, labels\n    \n    def transform(self, data_dir: str) -> pd.DataFrame:\n        \"\"\"\n        Transform test data using fitted pipeline (optimized)\n        \n        Args:\n            data_dir: Path to test data directory\n            \n        Returns:\n            Features DataFrame\n        \"\"\"\n        print(f\"Extracting features from test data: {data_dir}\")\n        \n        metadata_path = os.path.join(data_dir, 'metadata.csv')\n        \n        all_features = []\n        repertoire_ids = []\n        \n        # Use parallel processing\n        from joblib import Parallel, delayed\n        \n        def process_repertoire(file_path, rep_id):\n            \"\"\"Process single repertoire\"\"\"\n            repertoire_df = pd.read_csv(file_path, sep='\\t')\n            features = self.extract_repertoire_features(repertoire_df)\n            return features, rep_id\n        \n        if os.path.exists(metadata_path):\n            metadata = pd.read_csv(metadata_path)\n            \n            results = Parallel(n_jobs=self.n_jobs if hasattr(self, 'n_jobs') else 4)(\n                delayed(process_repertoire)(os.path.join(data_dir, row['filename']), row['repertoire_id'])\n                for _, row in tqdm(metadata.iterrows(), total=len(metadata), desc=\"Processing repertoires\")\n            )\n        else:\n            tsv_files = glob.glob(os.path.join(data_dir, '*.tsv'))\n            \n            results = Parallel(n_jobs=self.n_jobs if hasattr(self, 'n_jobs') else 4)(\n                delayed(process_repertoire)(file_path, os.path.basename(file_path).replace('.tsv', ''))\n                for file_path in tqdm(tsv_files, desc=\"Processing repertoires\")\n            )\n        \n        # Unpack results\n        for features, rep_id in results:\n            all_features.append(features)\n            repertoire_ids.append(rep_id)\n        \n        # Convert to DataFrame\n        features_df = pd.DataFrame(all_features)\n        features_df.index = repertoire_ids\n        \n        # Select only top features\n        for col in features_df.columns:\n            if col.startswith('kmer_') and col not in self.top_kmers:\n                features_df = features_df.drop(columns=[col])\n            elif col.startswith('vj_') and col not in self.top_vj:\n                features_df = features_df.drop(columns=[col])\n        \n        # Align with training features\n        for col in self.feature_names:\n            if col not in features_df.columns and not col.startswith('pca_'):\n                features_df[col] = 0\n        \n        if not self.use_pca:\n            features_df = features_df[self.feature_names]\n        \n        features_df = features_df.fillna(0)\n        \n        # Apply transformation\n        if self.use_pca:\n            # First scale, then PCA\n            temp_scaler = StandardScaler()\n            features_scaled = temp_scaler.fit_transform(features_df)\n            features_pca = self.pca.transform(features_scaled)\n            features_df = pd.DataFrame(features_pca, columns=self.feature_names, index=features_df.index)\n        else:\n            features_scaled = self.scaler.transform(features_df)\n            features_df = pd.DataFrame(features_scaled, columns=features_df.columns, index=features_df.index)\n        \n        return features_df\n\n## Optimized Model Trainer with Early Stopping and Reduced CV\n\nclass OptimizedModelTrainer(ModelTrainer):\n    \"\"\"\n    Optimized model trainer with faster training\n    \"\"\"\n    \n    def __init__(self, model_type: str = 'xgboost', n_jobs: int = 4, device: str = 'cpu',\n                 fast_mode: bool = True):\n        \"\"\"\n        Initialize optimized model trainer\n        \n        Args:\n            model_type: Type of model\n            n_jobs: Number of parallel jobs\n            device: Device to use\n            fast_mode: Whether to use faster hyperparameters\n        \"\"\"\n        super().__init__(model_type, n_jobs, device)\n        self.fast_mode = fast_mode\n    \n    def create_model(self):\n        \"\"\"Create model with optimized hyperparameters for speed\"\"\"\n        \n        if self.model_type == 'xgboost':\n            if self.fast_mode:\n                # Faster XGBoost settings\n                self.model = xgb.XGBClassifier(\n                    n_estimators=100,  # Reduced from 200\n                    max_depth=4,       # Reduced from 6\n                    learning_rate=0.1,  # Increased from 0.05\n                    subsample=0.8,\n                    colsample_bytree=0.8,\n                    random_state=42,\n                    n_jobs=self.n_jobs,\n                    tree_method='hist',  # Faster than exact\n                    max_bin=256,  # Reduced from default\n                    eval_metric='logloss'\n                )\n            else:\n                # Standard settings\n                self.model = xgb.XGBClassifier(\n                    n_estimators=200,\n                    max_depth=6,\n                    learning_rate=0.05,\n                    subsample=0.8,\n                    colsample_bytree=0.8,\n                    random_state=42,\n                    n_jobs=self.n_jobs,\n                    tree_method='hist',\n                    eval_metric='logloss'\n                )\n        \n        elif self.model_type == 'lightgbm':\n            if self.fast_mode:\n                # Faster LightGBM settings\n                self.model = lgb.LGBMClassifier(\n                    n_estimators=100,\n                    max_depth=4,\n                    learning_rate=0.1,\n                    subsample=0.8,\n                    colsample_bytree=0.8,\n                    random_state=42,\n                    n_jobs=self.n_jobs,\n                    num_leaves=31,  # Reduced\n                    verbose=-1\n                )\n            else:\n                self.model = lgb.LGBMClassifier(\n                    n_estimators=200,\n                    max_depth=6,\n                    learning_rate=0.05,\n                    subsample=0.8,\n                    colsample_bytree=0.8,\n                    random_state=42,\n                    n_jobs=self.n_jobs,\n                    verbose=-1\n                )\n        \n        elif self.model_type == 'rf':\n            if self.fast_mode:\n                # Faster Random Forest\n                self.model = RandomForestClassifier(\n                    n_estimators=50,  # Reduced from 200\n                    max_depth=8,      # Reduced from 10\n                    min_samples_split=10,  # Increased from 5\n                    min_samples_leaf=4,    # Increased from 2\n                    random_state=42,\n                    n_jobs=self.n_jobs,\n                    max_features='sqrt'  # Faster than 'auto'\n                )\n            else:\n                self.model = RandomForestClassifier(\n                    n_estimators=200,\n                    max_depth=10,\n                    min_samples_split=5,\n                    min_samples_leaf=2,\n                    random_state=42,\n                    n_jobs=self.n_jobs\n                )\n        \n        elif self.model_type == 'lr':\n            # Logistic Regression is already fast\n            self.model = LogisticRegression(\n                C=1.0,\n                max_iter=500,  # Reduced from 1000\n                random_state=42,\n                n_jobs=self.n_jobs,\n                solver='saga'  # Faster for large datasets\n            )\n        \n        else:\n            raise ValueError(f\"Unknown model type: {self.model_type}\")\n        \n        return self.model\n    \n    def cross_validate(self, X: pd.DataFrame, y: pd.Series, cv: int = 3):\n        \"\"\"\n        Perform cross-validation with reduced folds for speed\n        \n        Args:\n            X: Features\n            y: Labels\n            cv: Number of folds (reduced to 3 for speed)\n            \n        Returns:\n            Cross-validation scores\n        \"\"\"\n        print(f\"Performing {cv}-fold cross-validation...\")\n        \n        if self.model is None:\n            self.create_model()\n        \n        skf = StratifiedKFold(n_splits=cv, shuffle=True, random_state=42)\n        \n        scores = cross_val_score(\n            self.model, X, y,\n            cv=skf,\n            scoring='roc_auc',\n            n_jobs=1,  # Set to 1 to avoid nested parallelism\n            verbose=0\n        )\n        \n        print(f\"Cross-validation AUC: {scores.mean():.4f} (+/- {scores.std():.4f})\")\n        \n        return scores\n\n## Fast Important Sequences Identifier\n\nclass FastImportantSequencesIdentifier(ImportantSequencesIdentifier):\n    \"\"\"\n    Faster version of important sequences identifier\n    \"\"\"\n    \n    def identify_frequency_based(self, data_dir: str, top_k: int = 50000, \n                                 sample_size: int = None) -> pd.DataFrame:\n        \"\"\"\n        Identify important sequences with optional sampling for speed\n        \n        Args:\n            data_dir: Path to training data directory\n            top_k: Number of top sequences to return\n            sample_size: Maximum number of sequences to process per repertoire (None = all)\n            \n        Returns:\n            DataFrame with important sequences\n        \"\"\"\n        print(\"Identifying important sequences (frequency-based, optimized)...\")\n        \n        metadata = pd.read_csv(os.path.join(data_dir, 'metadata.csv'))\n        \n        # Collect sequences by label\n        pos_sequences = Counter()\n        neg_sequences = Counter()\n        \n        for idx, row in tqdm(metadata.iterrows(), total=len(metadata), desc=\"Processing repertoires\"):\n            file_path = os.path.join(data_dir, row['filename'])\n            repertoire_df = pd.read_csv(file_path, sep='\\t')\n            \n            sequences = repertoire_df['junction_aa'].dropna().tolist()\n            \n            # Sample if requested\n            if sample_size and len(sequences) > sample_size:\n                sequences = np.random.choice(sequences, sample_size, replace=False).tolist()\n            \n            if row['label_positive'] == 1:\n                pos_sequences.update(sequences)\n            else:\n                neg_sequences.update(sequences)\n        \n        # Calculate enrichment scores (vectorized)\n        all_sequences = set(pos_sequences.keys()) | set(neg_sequences.keys())\n        \n        # Pre-calculate totals\n        pos_total = sum(pos_sequences.values())\n        neg_total = sum(neg_sequences.values())\n        \n        sequence_scores = []\n        \n        # Batch processing\n        for seq in tqdm(all_sequences, desc=\"Calculating enrichment scores\"):\n            pos_count = pos_sequences[seq]\n            neg_count = neg_sequences[seq]\n            total = pos_count + neg_count\n            \n            if total >= 5:  # Minimum count threshold\n                pos_freq = pos_count / pos_total if pos_total > 0 else 0\n                neg_freq = neg_count / neg_total if neg_total > 0 else 0\n                \n                if neg_freq > 0:\n                    log_odds = np.log2((pos_freq + 1e-10) / (neg_freq + 1e-10))\n                else:\n                    log_odds = 10\n                \n                importance = abs(log_odds) * np.log(total + 1)\n                \n                sequence_scores.append({\n                    'junction_aa': seq,\n                    'pos_count': pos_count,\n                    'neg_count': neg_count,\n                    'log_odds_ratio': log_odds,\n                    'importance_score': importance\n                })\n        \n        # Sort and take top k\n        scores_df = pd.DataFrame(sequence_scores).sort_values('importance_score', ascending=False)\n        top_sequences = scores_df.head(top_k)\n        \n        # Add placeholder V and J genes\n        top_sequences['v_call'] = 'TRBV20-1'\n        top_sequences['j_call'] = 'TRBJ2-7'\n        \n        print(f\"Identified {len(top_sequences)} important sequences\")\n        \n        return top_sequences\n\n## Updated ImmuneStatePredictor with Optimizations\n\nclass OptimizedImmuneStatePredictor(ImmuneStatePredictor):\n    \"\"\"\n    Optimized immune state predictor with faster training\n    \"\"\"\n    \n    def __init__(self, n_jobs: int = 1, device: str = 'cpu', \n                 model_type: str = 'xgboost',\n                 kmer_sizes: List[int] = [3, 4],  # Reduced from [3,4,5]\n                 importance_method: str = 'frequency_based',\n                 fast_mode: bool = True,\n                 max_kmer_features: int = 1000,\n                 use_pca: bool = False,\n                 cv_folds: int = 3,  # Reduced from 5\n                 **kwargs):\n        \"\"\"\n        Initialize optimized predictor\n        \n        Args:\n            n_jobs: Number of CPU cores\n            device: Device to use\n            model_type: Type of model to use\n            kmer_sizes: K-mer sizes for feature extraction\n            importance_method: Method for identifying important sequences\n            fast_mode: Whether to use faster settings\n            max_kmer_features: Maximum number of k-mer features\n            use_pca: Whether to use PCA\n            cv_folds: Number of cross-validation folds\n        \"\"\"\n        total_cores = os.cpu_count()\n        if n_jobs == -1:\n            self.n_jobs = total_cores\n        else:\n            self.n_jobs = min(n_jobs, total_cores)\n        \n        self.device = device\n        self.model_type = model_type\n        self.kmer_sizes = kmer_sizes\n        self.importance_method = importance_method\n        self.fast_mode = fast_mode\n        self.cv_folds = cv_folds\n        \n        # Initialize optimized components\n        self.feature_eng = OptimizedFeatureEngineering(\n            kmer_sizes=kmer_sizes,\n            use_vj_features=True,\n            use_physicochemical=True,\n            use_diversity=True,\n            max_kmer_features=max_kmer_features,\n            use_pca=use_pca\n        )\n        self.feature_eng.n_jobs = self.n_jobs\n        \n        self.model_trainer = OptimizedModelTrainer(\n            model_type=model_type,\n            n_jobs=self.n_jobs,\n            device=self.device,\n            fast_mode=fast_mode\n        )\n        \n        self.seq_identifier = FastImportantSequencesIdentifier(\n            method=importance_method\n        )\n        \n        self.important_sequences_ = None\n        self.model = None\n    \n    def fit(self, train_dir_path: str):\n        \"\"\"\n        Train the model on training data (optimized)\n        \n        Args:\n            train_dir_path: Path to training data directory\n            \n        Returns:\n            self\n        \"\"\"\n        print(f\"\\n{'='*80}\")\n        print(f\"TRAINING IMMUNE STATE PREDICTOR (OPTIMIZED)\")\n        print(f\"{'='*80}\\n\")\n        \n        # Step 1: Feature extraction\n        print(\"Step 1: Feature Extraction (Optimized)\")\n        start_time = time.time()\n        X_train, y_train = self.feature_eng.fit_transform(train_dir_path)\n        print(f\"Feature extraction time: {time.time() - start_time:.2f}s\")\n        \n        # Step 2: Model training\n        print(\"\\nStep 2: Model Training (Optimized)\")\n        start_time = time.time()\n        self.model_trainer.train(X_train, y_train)\n        self.model = self.model_trainer.model\n        print(f\"Training time: {time.time() - start_time:.2f}s\")\n        \n        # Step 3: Cross-validation (optional, can be skipped for speed)\n        if self.cv_folds > 0:\n            print(f\"\\nStep 3: Cross-Validation ({self.cv_folds} folds)\")\n            start_time = time.time()\n            cv_scores = self.model_trainer.cross_validate(X_train, y_train, cv=self.cv_folds)\n            print(f\"Cross-validation time: {time.time() - start_time:.2f}s\")\n        \n        # Step 4: Identify important sequences\n        print(\"\\nStep 4: Identifying Important Sequences (Optimized)\")\n        start_time = time.time()\n        self.important_sequences_ = self.seq_identifier.identify_frequency_based(\n            train_dir_path,\n            top_k=50000,\n            sample_size=5000 if self.fast_mode else None  # Sample for speed\n        )\n        print(f\"Sequence identification time: {time.time() - start_time:.2f}s\")\n        \n        print(f\"\\n{'='*80}\")\n        print(\"TRAINING COMPLETE!\")\n        print(f\"{'='*80}\\n\")\n        \n        return self\n\n## Usage Example with Timing\n\nimport time\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"STARTING OPTIMIZED IMMUNE STATE PREDICTION PIPELINE\")\nprint(\"=\"*80 + \"\\n\")\n\n# Process each dataset pair\nfor train_dir, test_dirs in train_test_dataset_pairs:\n    print(f\"\\n{'='*80}\")\n    print(f\"Processing: {os.path.basename(train_dir)}\")\n    print(f\"{'='*80}\\n\")\n    \n    total_start = time.time()\n    \n    # Initialize optimized predictor\n    predictor = OptimizedImmuneStatePredictor(\n        n_jobs=-1,  # Use all available cores\n        device='cpu',\n        model_type='xgboost',  # Or 'lightgbm' for even faster training\n        kmer_sizes=[3, 4],  # Reduced from [3,4,5]\n        importance_method='frequency_based',\n        fast_mode=True,  # Enable fast mode\n        max_kmer_features=1000,  # Limit k-mer features\n        use_pca=False,  # Set to True for even more speed (but may reduce accuracy)\n        cv_folds=3  # Reduced from 5\n    )\n    \n    # Train\n    predictor.fit(train_dir)\n    \n    # Save model\n    model_save_dir = os.path.join(results_dir, 'models', os.path.basename(train_dir))\n    predictor.save(model_save_dir)\n    \n    # Predict on test sets\n    for test_dir in test_dirs:\n        predictions = predictor.predict_proba(test_dir)\n        \n        # Save predictions\n        pred_path = os.path.join(\n            results_dir,\n            f\"{os.path.basename(train_dir)}_test_predictions.tsv\"\n        )\n        save_tsv(predictions, pred_path)\n    \n    # Save important sequences\n    important_seqs = predictor.identify_associated_sequences(\n        dataset_name=os.path.basename(train_dir),\n        top_k=50000\n    )\n    \n    seq_path = os.path.join(\n        results_dir,\n        f\"{os.path.basename(train_dir)}_important_sequences.tsv\"\n    )\n    save_tsv(important_seqs, seq_path)\n    \n    total_time = time.time() - total_start\n    print(f\"\\nTotal processing time for {os.path.basename(train_dir)}: {total_time:.2f}s ({total_time/60:.2f} min)\")\n\n# Concatenate all outputs\nconcatenate_output_files(out_dir=results_dir)\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"OPTIMIZED PIPELINE COMPLETE!\")\nprint(\"=\"*80)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T10:07:07.636941Z","iopub.execute_input":"2025-11-06T10:07:07.637321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null}]}