{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"import os\n\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T17:56:32.233336Z","iopub.execute_input":"2025-08-12T17:56:32.233601Z","iopub.status.idle":"2025-08-12T17:56:32.243318Z","shell.execute_reply.started":"2025-08-12T17:56:32.233551Z","shell.execute_reply":"2025-08-12T17:56:32.242346Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Task 1: Extrapolatory Data Analysis","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\n\n# Set the correct file path\ndata_path = '/kaggle/input/leash-BELKA/train.csv'\n\n# Load a subset of the data for faster processing\ndf_train = pd.read_csv(data_path, nrows=100000)\n\ntest_path = '/kaggle/input/leash-BELKA/test.csv'\ndf_test = pd.read_csv(test_path)\n\nprint(\"Dataset loaded successfully!\")\n\n# Data Inspection\nprint(\"\\nFirst 5 rows of the dataframe:\")\nprint(df_train.head())\n\nprint(\"\\nSummary information:\")\ndf_train.info()\n\nprint(\"\\nDescriptive statistics:\")\nprint(df_train.describe())\n\n# Target Variable Analysis\nprint(\"\\nDistribution of the target variable `binds`:\")\nprint(df_train['binds'].value_counts())\n\nplt.figure(figsize=(6, 4))\nsns.countplot(x='binds', data=df_train)\nplt.title('Distribution of the Target Variable (binds)')\nplt.xlabel('Binds (0 = No, 1 = Yes)')\nplt.ylabel('Count')\nplt.show()\n\n# Feature Analysis\nprint(\"\\nMissing values in each column:\")\nprint(df_train.isnull().sum())\n\nnum_unique_molecules = df_train['molecule_smiles'].nunique()\nprint(f\"\\nNumber of unique molecules in the subset: {num_unique_molecules}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:15:59.014864Z","iopub.execute_input":"2025-08-12T18:15:59.015127Z","iopub.status.idle":"2025-08-12T18:16:05.689412Z","shell.execute_reply.started":"2025-08-12T18:15:59.015107Z","shell.execute_reply":"2025-08-12T18:16:05.688601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate and print the percentage of each class to highlight imbalance\ntotal_rows = len(df_train)\nbinds_counts = df_train['binds'].value_counts()\npercentage_binds_1 = (binds_counts.get(1, 0) / total_rows) * 100\npercentage_binds_0 = (binds_counts.get(0, 0) / total_rows) * 100\n\nprint(f\"\\nPercentage of 'binds=1' (positive class): {percentage_binds_1:.2f}%\")\nprint(f\"Percentage of 'binds=0' (negative class): {percentage_binds_0:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:16:27.495814Z","iopub.execute_input":"2025-08-12T18:16:27.496572Z","iopub.status.idle":"2025-08-12T18:16:27.502757Z","shell.execute_reply.started":"2025-08-12T18:16:27.496524Z","shell.execute_reply":"2025-08-12T18:16:27.501857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze the 'protein_name' feature\nprint(\"\\nUnique protein names and their counts:\")\nprotein_counts = df_train['protein_name'].value_counts()\nprint(protein_counts)\n\n# Visualize the distribution of protein names\nplt.figure(figsize=(10, 6))\nsns.countplot(y='protein_name', data=df_train, order=protein_counts.index)\nplt.title('Distribution of Protein Names')\nplt.xlabel('Count')\nplt.ylabel('Protein Name')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:16:36.272010Z","iopub.execute_input":"2025-08-12T18:16:36.272605Z","iopub.status.idle":"2025-08-12T18:16:36.459625Z","shell.execute_reply.started":"2025-08-12T18:16:36.272577Z","shell.execute_reply":"2025-08-12T18:16:36.458991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create new features for the length of SMILES strings\ndf_train['molecule_smiles_length'] = df_train['molecule_smiles'].apply(len)\ndf_train['buildingblock1_smiles_length'] = df_train['buildingblock1_smiles'].apply(len)\n\n# Analyze the distribution of these new features\nprint(\"\\nDescriptive statistics for molecule and building block SMILES lengths:\")\nprint(df_train[['molecule_smiles_length', 'buildingblock1_smiles_length']].describe())\n\n# Visualize the distribution of molecule SMILES length\nplt.figure(figsize=(10, 5))\nsns.histplot(df_train['molecule_smiles_length'], bins=50, kde=True)\nplt.title('Distribution of Molecule SMILES Lengths')\nplt.xlabel('Molecule SMILES Length')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:16:53.196203Z","iopub.execute_input":"2025-08-12T18:16:53.196811Z","iopub.status.idle":"2025-08-12T18:16:54.033008Z","shell.execute_reply.started":"2025-08-12T18:16:53.196785Z","shell.execute_reply":"2025-08-12T18:16:54.032396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compare molecule length for positive and negative binding events\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='binds', y='molecule_smiles_length', data=df_train)\nplt.title('Molecule SMILES Length vs. Binding Status')\nplt.xlabel('Binds (0 = No, 1 = Yes)')\nplt.ylabel('Molecule SMILES Length')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:17:14.666195Z","iopub.execute_input":"2025-08-12T18:17:14.666461Z","iopub.status.idle":"2025-08-12T18:17:14.823301Z","shell.execute_reply.started":"2025-08-12T18:17:14.666442Z","shell.execute_reply":"2025-08-12T18:17:14.822724Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Task 2 : Data Preprocessing","metadata":{}},{"cell_type":"code","source":"!pip install rdkit-pypi\n!pip install pandarallel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:17:28.319010Z","iopub.execute_input":"2025-08-12T18:17:28.319611Z","iopub.status.idle":"2025-08-12T18:17:41.064073Z","shell.execute_reply.started":"2025-08-12T18:17:28.319585Z","shell.execute_reply":"2025-08-12T18:17:41.063283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Task 2: Data Preprocessing\n\nimport pandas as pd\nfrom pandarallel import pandarallel\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\nimport numpy as np\n\n# Initialize pandarallel\npandarallel.initialize(progress_bar=True, verbose=0)\n\n# Take a small sample\ndf_train_subset = df_train.sample(n=30000, random_state=42)\ndf_test_subset = pd.read_csv('/kaggle/input/leash-BELKA/test.csv', nrows=10000)\n\n# Fingerprint function\ndef smiles_to_fingerprint(smiles_string):\n    mol = Chem.MolFromSmiles(smiles_string)\n    if mol is not None:\n        return np.array(AllChem.GetMorganFingerprintAsBitVect(mol, 2, nBits=1024))\n    else:\n        return np.zeros(1024, dtype=int)\n\n# Generate fingerprints\ntrain_fp_series = df_train_subset['molecule_smiles'].parallel_apply(smiles_to_fingerprint)\ntest_fp_series = df_test_subset['molecule_smiles'].parallel_apply(smiles_to_fingerprint)\n\n# Convert to DataFrames, now with explicit STRING column names\nfp_cols = [f'fp_{i}' for i in range(1024)] # Create string names like 'fp_0', 'fp_1', etc.\nX_train_fp = pd.DataFrame(train_fp_series.to_list(), columns=fp_cols).reset_index(drop=True)\nX_test_fp = pd.DataFrame(test_fp_series.to_list(), columns=fp_cols).reset_index(drop=True)\n\n# Reset the index on the one-hot encoded data\nX_train_protein = pd.get_dummies(df_train_subset['protein_name'], prefix='protein').reset_index(drop=True)\nX_test_protein = pd.get_dummies(df_test_subset['protein_name'], prefix='protein').reset_index(drop=True)\n\n# Concatenation \nX_train = pd.concat([X_train_protein, X_train_fp], axis=1)\nX_test = pd.concat([X_test_protein, X_test_fp], axis=1)\n\n# The target 'y' must also be reset to match\ny = df_train_subset['binds'].reset_index(drop=True)\n\n# Align columns and finish\nX_train, X_test = X_train.align(X_test, join='left', axis=1, fill_value=0)\ndf_test = df_test_subset\n\nprint(\"\\n Data preprocessing complete!\")\nprint(f\"All {len(X_train.columns)} column names are now strings.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:30:37.848450Z","iopub.execute_input":"2025-08-12T18:30:37.848736Z","iopub.status.idle":"2025-08-12T18:32:11.468439Z","shell.execute_reply.started":"2025-08-12T18:30:37.848715Z","shell.execute_reply":"2025-08-12T18:32:11.467656Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Task 3 & 4: Model & Training/Test ","metadata":{}},{"cell_type":"code","source":"# Task 3 & 4: Model, Train, and Evaluate\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import classification_report, roc_auc_score\n\n# 1. Split the data into a training set and a validation set\n# We use the X_train and y data we created during preprocessing\n# The model will train on the 'tr' set and we'll check it on the 'val' set\nX_tr, X_val, y_tr, y_val = train_test_split(X_train, y, test_size=0.2, random_state=42, stratify=y)\n\n# 2. Initialize the Logistic Regression model\n# class_weight='balanced' helps the model handle the highly imbalanced data\nmodel = LogisticRegression(class_weight='balanced', solver='liblinear', random_state=42, max_iter=1000)\n\n# 3. Train the model on the training portion of the data\nprint(\"Training the model...\")\nmodel.fit(X_tr, y_tr)\nprint(\"Model training complete.\")\n\n# 4. Evaluate the model on the unseen validation set\nprint(\"\\n Model Evaluation on Validation Set \")\n# Predict probabilities for the positive class (binds=1)\ny_pred_proba = model.predict_proba(X_val)[:, 1]\n\n# To generate a classification report, we need class labels (0 or 1)\ny_pred_class = model.predict(X_val)\n\n# Print the performance metrics\nroc_auc = roc_auc_score(y_val, y_pred_proba)\nprint(f\"ROC AUC Score: {roc_auc:.4f}\")\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_val, y_pred_class))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:32:36.794991Z","iopub.execute_input":"2025-08-12T18:32:36.795282Z","iopub.status.idle":"2025-08-12T18:32:38.641247Z","shell.execute_reply.started":"2025-08-12T18:32:36.795257Z","shell.execute_reply":"2025-08-12T18:32:38.640541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"\nprint(\"Making predictions on the official test data...\")\n\n# Use the trained model to predict the binding probabilities on the test set\n# We use .predict_proba() because the competition metric (average precision) uses probabilities.\ntest_predictions = model.predict_proba(X_test)[:, 1]\n\nprint(\"Predictions generated successfully.\")\n\n# Create the final submission file in the required format\nsubmission_df = pd.DataFrame({'id': df_test['id'], 'binds': test_predictions})\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"\\nSubmission file 'submission.csv' created successfully.\")\nprint(\"First 5 rows of the submission file:\")\nprint(submission_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T18:32:49.680815Z","iopub.execute_input":"2025-08-12T18:32:49.681058Z","iopub.status.idle":"2025-08-12T18:32:49.819146Z","shell.execute_reply.started":"2025-08-12T18:32:49.681042Z","shell.execute_reply":"2025-08-12T18:32:49.818366Z"}},"outputs":[],"execution_count":null}]}