{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":106680,"databundleVersionId":13374319,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nimport seaborn as sns\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport xgboost as xgb\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import log_loss\nfrom sklearn.metrics import roc_auc_score\n\nimport matplotlib.pyplot as plt\nfrom typing import List, Optional\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.decomposition import PCA # Import PCA\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import log_loss\nfrom sklearn.metrics import roc_auc_score\nfrom xgboost import plot_importance\n\nimport optuna","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Training Dataset (EDA & Visualization)**","metadata":{}},{"cell_type":"code","source":"PATH_DATASET = \"/kaggle/input/adaptive-immune-profiling-challenge-2025\"\nPATH_TRAIN_DATASETS = os.path.join(PATH_DATASET, 'train_datasets', 'train_datasets')\ntrain_datasets = sorted(os.listdir(PATH_TRAIN_DATASETS))\nprint(train_datasets)\nPATH_TEST_DATASETS = os.path.join(PATH_DATASET, 'test_datasets', 'test_datasets')\ntest_datasets = sorted(os.listdir(PATH_TEST_DATASETS))\nprint(test_datasets)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_tsv_files_export_parquet(folder_path: str, output_path: str, show_hist: Optional[List[str]] = None):\n    folder = os.path.basename(folder_path) # Derive folder name from folder_path\n    # List all files in the directory\n    files = os.listdir(folder_path)\n\n    # Filter for .tsv files\n    tsv_files = [f for f in files if f.endswith('.tsv')]\n    other_files = [f.name for f in os.scandir(folder_path) if not f.name.endswith('.tsv')]\n    print(f'Loading {len(tsv_files)} .tsv files from {folder} (remaining: {other_files}).')\n\n    # Iterate through each TSV file, load it into a DataFrame, and print column names\n    dfs = []\n    for tsv_file in tqdm(tsv_files, desc=\"Loading TSV files\"):\n        file_path = os.path.join(folder_path, tsv_file)\n        file_name, _ = os.path.splitext(tsv_file)\n        try:\n            df = pd.read_csv(file_path, sep='\\t')\n            df['repertoire_id'] = file_name\n            dfs.append(df)\n        except Exception as e:\n            print(f\"Error loading {tsv_file}: {e}\")\n\n    merged_df = pd.concat(dfs, ignore_index=True)\n    del dfs # Free up memory\n\n    print(f\"Merged DataFrame shape: {merged_df.shape}\")\n    for col in merged_df.columns:\n        print(f\"Unique values in column '{col}': {len(merged_df[col].unique())}\")\n    print(\"Merged DataFrame head:\")\n    display(merged_df.head())\n\n    os.makedirs(output_path, exist_ok=True)\n    merged_df.to_parquet(f'{output_path}/{folder}.parquet')\n\n    # Plot histograms for specified columns if show_hist is provided\n    if not isinstance(show_hist, list) and not show_hist:\n        return\n    print(f\"Plotting histograms for columns: {', '.join(show_hist)}\")\n    for col in show_hist:\n        if col not in merged_df.columns:\n            print(f\"Warning: Column '{col}' not found in the DataFrame for {folder}.\")\n            continue\n        # Get all value counts\n        all_counts = merged_df[col].value_counts()\n        \n        plt.figure(figsize=(min(12, len(all_counts) * 0.3), 4)) # Adjust figure size dynamically\n        sns.barplot(x=all_counts.index, y=all_counts.values, palette='viridis')\n        plt.title(f'Value Counts for {col} in {folder}')\n        plt.xlabel(col)\n        plt.ylabel('Count')\n        plt.xticks(rotation=90, ha='center') # Rotate labels more for many categories\n        plt.grid(True)\n        plt.tight_layout()\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Iterate over all sub-datasets\nfor folder in tqdm(train_datasets):\n    path_dataset_ = os.path.join(PATH_TRAIN_DATASETS, folder)\n    load_tsv_files_export_parquet(\n        path_dataset_, output_path='train_dataset', show_hist=['v_call', 'j_call', 'd_call'])\n    new_meta_csv = os.path.join(\"train_dataset\", f\"{folder}-metadata.csv\")\n    shutil.copy(os.path.join(path_dataset_, \"metadata.csv\"), new_meta_csv)\n    df_meta = pd.read_csv(new_meta_csv)\n    display(df_meta)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Testing Dataset (EDA & Visualization)**","metadata":{}},{"cell_type":"code","source":"# Iterate over all sub-datasets\nfor folder in tqdm(test_datasets):\n    path_dataset_ = os.path.join(PATH_TEST_DATASETS, folder)\n    load_tsv_files_export_parquet(\n        path_dataset_, output_path='test_dataset', show_hist=['v_call', 'j_call', 'd_call'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **XGBOOST MODEL**","metadata":{}},{"cell_type":"code","source":"# Define function to parse a dataset folder of TSV files\ndef parse_tsv_files(folder_path, feature_columns=('v_call', 'j_call')):\n    folder_name = os.path.basename(folder_path)\n    files = os.listdir(folder_path)\n    tsv_files = [f for f in files if f.endswith('.tsv')]\n    other_files = [f.name for f in os.scandir(folder_path) if not f.name.endswith('.tsv')]\n\n    print(f'Loading {len(tsv_files)} .tsv files from {folder_name} (remaining: {other_files}).')\n\n    metadata = None\n    if \"metadata.csv\" in files:\n        metadata = pd.read_csv(os.path.join(folder_path, \"metadata.csv\"))\n        metadata.set_index(\"filename\", inplace=True)\n\n    dataset_rows = []\n\n    for tsv_file in tqdm(tsv_files, desc=f\"Loading {folder_name}\"):\n        file_path = os.path.join(folder_path, tsv_file)\n        file_name, _ = os.path.splitext(tsv_file)\n\n        try:\n            df = pd.read_csv(file_path, sep=\"\\t\")\n        except Exception as e:\n            print(f\"Error loading {tsv_file}: {e}\")\n            continue\n\n        row = {\n            \"ID\": file_name,\n            \"dataset\": folder_name\n        }\n\n        if metadata is not None:\n            row[\"label_positive\"] = int(metadata.at[tsv_file, \"label_positive\"])\n\n        for col in feature_columns:\n            counts = df[col].value_counts() / len(df)\n            row.update(counts.to_dict())\n\n        dataset_rows.append(row)\n\n    return dataset_rows","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to load all train and test folders\ndef load_all_datasets(train_folders, test_folders, train_base_path, test_base_path):\n    train_rows = []\n    test_rows = []\n\n    for folder in tqdm(train_folders, desc=\"Train folders\"):\n        path = os.path.join(train_base_path, folder)\n        train_rows += parse_tsv_files(path)\n\n    for folder in tqdm(test_folders, desc=\"Test folders\"):\n        path = os.path.join(test_base_path, folder)\n        test_rows += parse_tsv_files(path)\n\n    return pd.DataFrame(train_rows), pd.DataFrame(test_rows)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to run XGB Grid Search\ndef run_grid_search_xgb(X_train, y_train, base_params, param_grid, n_splits):\n    cv_splitter = StratifiedKFold(\n        n_splits=n_splits,\n        shuffle=True,\n        random_state=42\n    )\n\n    model = xgb.XGBClassifier(\n        **base_params,\n        use_label_encoder=False\n    )\n\n    grid_search = GridSearchCV(\n        estimator=model,\n        param_grid=param_grid,\n        scoring='neg_log_loss',\n        cv=cv_splitter,\n        n_jobs=-1,\n        verbose=2,\n        refit=True\n    )\n\n    print(\"\\nRunning Grid Search...\")\n    grid_search.fit(X_train, y_train)\n\n    print(\"Best Params:\", grid_search.best_params_)\n    print(\"Best Neg LogLoss:\", grid_search.best_score_)\n\n    return grid_search.best_estimator_","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to train the best model on full data\ndef train_final_model(model, X_train, y_train):\n    print(\"\\nTraining final model...\")\n    model.fit(X_train, y_train)\n    return model","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to generate predictions\ndef generate_predictions(model, X_test, dataset_test):\n    print(\"\\nGenerating predictions...\")\n    probabilities = model.predict_proba(X_test)[:, 1]\n\n    predictions_df = pd.DataFrame({\n        \"ID\": dataset_test[\"ID\"],\n        \"dataset\": dataset_test[\"dataset\"],\n        \"label_positive_probability\": probabilities\n    })\n\n    return predictions_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to create submission.csv\ndef prepare_submission(predictions_df, sample_path, output_path):\n    sample_df = pd.read_csv(sample_path)\n    sample_df = sample_df.drop(columns=['label_positive_probability'])\n\n    merged = pd.merge(\n        sample_df,\n        predictions_df,\n        on=['ID', 'dataset'],\n        how='left'\n    )\n\n    merged = merged.fillna(0.5)\n    merged = merged.drop_duplicates(subset=['ID', 'dataset'], keep='first')\n\n    merged.to_csv(output_path, index=False)\n    print(f\"Saved submission to: {output_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Submission**","metadata":{}},{"cell_type":"code","source":"# Define main workflow\ndef main():\n    PATH_DATASET = \"/kaggle/input/adaptive-immune-profiling-challenge-2025\"\n    PATH_TRAIN_DATASETS = os.path.join(PATH_DATASET, 'train_datasets', 'train_datasets')\n    PATH_TEST_DATASETS = os.path.join(PATH_DATASET, 'test_datasets', 'test_datasets')\n\n    train_folders = sorted(os.listdir(PATH_TRAIN_DATASETS))\n    test_folders = sorted(os.listdir(PATH_TEST_DATASETS))\n\n    print(\"Loading Train/Test datasets...\")\n    dataset_train, dataset_test = load_all_datasets(\n        train_folders,\n        test_folders,\n        PATH_TRAIN_DATASETS,\n        PATH_TEST_DATASETS\n    )\n\n    X_train = dataset_train.drop(['label_positive', 'ID', 'dataset'], axis=1).fillna(0)\n    y_train = dataset_train['label_positive']\n    X_test = dataset_test.drop(['ID', 'dataset'], axis=1).fillna(0)\n\n    excluded_features = ['TCRBV6-01', 'TCRBVA']\n    X_train = X_train.drop(excluded_features, axis=1, errors='ignore')\n\n    for col in set(X_train.columns) - set(X_test.columns):\n        X_test[col] = 0\n\n    X_test = X_test[X_train.columns]\n\n    base_params = {\n        'eval_metric': 'logloss',\n        'objective': 'binary:logistic',\n        'random_state': 42,\n        'importance_type': 'gain'\n    }\n\n    param_grid = {\n        'colsample_bytree': [0.58, 0.63842335244, 0.70],\n        'learning_rate': [0.02, 0.03256233, 0.05],\n        'max_depth': [18, 20, 22],\n        'reg_alpha': [0.05, 0.087314708, 0.15],\n        'reg_lambda': [0.003, 0.0072173962, 0.02],\n        'subsample': [0.7, 0.8, 0.9]\n    }\n\n    best_model = run_grid_search_xgb(\n        X_train,\n        y_train,\n        base_params,\n        param_grid,\n        n_splits=8\n    )\n\n    final_model = train_final_model(\n        best_model,\n        X_train,\n        y_train\n    )\n\n    predictions = generate_predictions(\n        final_model,\n        X_test,\n        dataset_test\n    )\n\n    sample_path = os.path.join(PATH_DATASET, \"sample_submissions.csv\")\n    prepare_submission(\n        predictions,\n        sample_path,\n        \"submission.csv\"\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run main workflow\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}