{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":106680,"databundleVersionId":13374319,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import StackingClassifier \nfrom sklearn.linear_model import LogisticRegression\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\n\n# ===============================================================\n# --- 1. DUMMY DATA SETUP (REPLACE WITH YOUR REAL DATA) ---\n# This section defines X_train, y_train, and X_test to fix the NameError.\n# To compete, you must replace this with code that loads the AIRR-ML-25 files \n# and performs your best feature engineering (k-mer counts, gene frequencies, etc.)\n\nnum_samples_train = 1000  # Number of training samples\nnum_samples_test = 500    # Number of test samples\nnum_features = 50         # Number of features (columns)\n\n# Create placeholder feature matrices (X_train, X_test)\nX_train = pd.DataFrame(np.random.rand(num_samples_train, num_features))\ny_train = pd.Series(np.random.randint(0, 2, size=num_samples_train)) # Target: 0 (Healthy) or 1 (Disease)\nX_test = pd.DataFrame(np.random.rand(num_samples_test, num_features))\n\n# Create a placeholder submission DataFrame with a correct ID column\n# (You will need to load the actual test IDs from the competition file)\nsubmission_df = pd.DataFrame({'ID': [f'sample_{i}' for i in range(num_samples_test)]}) \n\nprint(\"Dummy data for X_train, y_train, and X_test has been successfully created.\")\n# --- END DUMMY DATA SETUP ---\n# ===============================================================\n\n\n# --- 2. Define Base Models (The First Layer) ---\n# These powerful models learn from your features independently.\n# NOTE: The optimal hyperparameters should be found via tuning (e.g., GridSearchCV).\nlgbm = LGBMClassifier(random_state=42, n_estimators=500, learning_rate=0.05, n_jobs=-1, verbose=-1)\nxgb = XGBClassifier(random_state=42, n_estimators=500, learning_rate=0.05, use_label_encoder=False, eval_metric='logloss', n_jobs=-1)\ncat = CatBoostClassifier(random_state=42, n_estimators=500, learning_rate=0.05, verbose=0, allow_writing_files=False)\n\nestimators = [\n    ('lgbm', lgbm),\n    ('xgb', xgb),\n    ('cat', cat)\n]\n\n# --- 3. Define the Stacking Classifier (The Second Layer) ---\n# The final_estimator combines the predictions of the three base models.\nstacking_model = StackingClassifier(\n    estimators=estimators,\n    final_estimator=LogisticRegression(solver='lbfgs', C=0.1, random_state=42),\n    cv=StratifiedKFold(n_splits=5, shuffle=True, random_state=42), \n    n_jobs=-1,\n    verbose=0\n)\n\n# --- 4. Train the Stacking Model ---\nprint(\"\\nStarting Stacking Model Training...\")\n# This line now works because X_train and y_train are defined above.\nstacking_model.fit(X_train, y_train) \nprint(\"Training Complete.\")\n\n# --- 5. Generate Predictions ---\n# We use predict_proba to get the probability of the positive class (Class 1 / Disease)\n# This is required for the AUC metric used in the competition.\npredictions_proba = stacking_model.predict_proba(X_test)[:, 1]\nprint(\"Predictions generated.\")\n\n# --- 6. Create Submission File ---\n# Add the predicted probabilities to your submission DataFrame\nsubmission_df['label_positive_probability'] = predictions_proba\n\n# Display the first few rows of the final submission (using dummy data)\nprint(\"\\nFirst 5 rows of the submission file:\")\nprint(submission_df.head())\n\n# To save the file for submission, uncomment the line below:\n# submission_df.to_csv('airr_ml_stacking_submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T18:54:39.452293Z","iopub.execute_input":"2025-12-14T18:54:39.452545Z","iopub.status.idle":"2025-12-14T18:55:09.722544Z","shell.execute_reply.started":"2025-12-14T18:54:39.452526Z","shell.execute_reply":"2025-12-14T18:55:09.722004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom tqdm.auto import tqdm # Import tqdm for progress bar (helps with large files)\n\ndef load_and_engineer_airr_data(data_path='../input/adaptive-immune-profiling-challenge-2025'):\n    \n    # 1. Load Metadata (Contains the 'filename' and the 'y' labels)\n    metadata_df = pd.read_csv(os.path.join(data_path, 'metadata.csv'))\n    \n    # Prepare the DataFrame to store the new features\n    feature_list = []\n    \n    # Define the folders where the TSV files are located\n    # NOTE: You MUST replace these paths with the actual folder names in your Kaggle/local directory\n    tsv_folders = ['train_repertoire', 'test_repertoire'] \n    \n    # Loop through all files mentioned in the metadata\n    for index, row in tqdm(metadata_df.iterrows(), total=len(metadata_df), desc=\"Extracting Features\"):\n        filename = row['filename']\n        \n        # Determine if the file is in the train or test folder\n        file_path = None\n        for folder in tsv_folders:\n            check_path = os.path.join(data_path, folder, filename)\n            if os.path.exists(check_path):\n                file_path = check_path\n                break\n        \n        if file_path is None:\n            # Handle files not found (e.g., test files you don't have the labels for yet)\n            continue\n            \n        # --- CORE FEATURE EXTRACTION ---\n        try:\n            # Read the TSV file using tab separator\n            df_repertoire = pd.read_csv(file_path, sep='\\t')\n            \n            # 1. V-Gene Frequency Feature\n            # Normalize=True calculates the relative frequency (percentage)\n            v_gene_counts = df_repertoire['v_call'].value_counts(normalize=True)\n            \n            # Find the frequency of the most common V-gene\n            max_v_freq = v_gene_counts.max() if not v_gene_counts.empty else 0.0\n            \n            # 2. Total Clones Feature (Counts are always useful)\n            total_clones = len(df_repertoire)\n            \n            # Create a dictionary for this patient's features\n            patient_features = {\n                'filename': filename,\n                'max_v_gene_freq': max_v_freq,\n                'total_clones': total_clones,\n                # ... Add more features here (e.g., max_j_gene_freq, unique_clone_count)\n            }\n            \n            feature_list.append(patient_features)\n            \n        except Exception as e:\n            # Handle bad files\n            print(f\"Error processing {filename}: {e}\")\n            pass\n            \n    # 3. Combine Features and Labels\n    feature_df = pd.DataFrame(feature_list)\n    final_df = metadata_df.merge(feature_df, on='filename', how='left')\n\n    # Prepare final X and y\n    X = final_df[['max_v_gene_freq', 'total_clones']] # Your real features\n    y = final_df['disease_label'].map({'healthy': 0, 'disease': 1})\n    \n    # --- IMPORTANT: Handle Missing Data ---\n    # Since you are only extracting a few simple features, some files might not have data.\n    # Impute NaNs with 0 or the mean.\n    X = X.fillna(0) \n\n    return X, y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T19:04:34.472929Z","iopub.execute_input":"2025-12-14T19:04:34.473183Z","iopub.status.idle":"2025-12-14T19:04:34.480935Z","shell.execute_reply.started":"2025-12-14T19:04:34.473166Z","shell.execute_reply":"2025-12-14T19:04:34.479819Z"}},"outputs":[],"execution_count":null}]}