{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"import os\n\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:24:47.522722Z","iopub.execute_input":"2025-08-12T16:24:47.523073Z","iopub.status.idle":"2025-08-12T16:24:47.534266Z","shell.execute_reply.started":"2025-08-12T16:24:47.52305Z","shell.execute_reply":"2025-08-12T16:24:47.532926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Task 1: Extrapolatory Data Analysis","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set the correct file path\ndata_path = '/kaggle/input/leash-BELKA/train.csv'\n\n# Load a subset of the data for faster processing\ndf_train = pd.read_csv(data_path, nrows=100000)\n\nprint(\"Dataset loaded successfully!\")\n\n# Data Inspection\nprint(\"\\nFirst 5 rows of the dataframe:\")\nprint(df_train.head())\n\nprint(\"\\nSummary information:\")\ndf_train.info()\n\nprint(\"\\nDescriptive statistics:\")\nprint(df_train.describe())\n\n# Target Variable Analysis\nprint(\"\\nDistribution of the target variable `binds`:\")\nprint(df_train['binds'].value_counts())\n\nplt.figure(figsize=(6, 4))\nsns.countplot(x='binds', data=df_train)\nplt.title('Distribution of the Target Variable (binds)')\nplt.xlabel('Binds (0 = No, 1 = Yes)')\nplt.ylabel('Count')\nplt.show()\n\n# Feature Analysis\nprint(\"\\nMissing values in each column:\")\nprint(df_train.isnull().sum())\n\nnum_unique_molecules = df_train['molecule_smiles'].nunique()\nprint(f\"\\nNumber of unique molecules in the subset: {num_unique_molecules}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:14:03.1772Z","iopub.execute_input":"2025-08-12T16:14:03.177885Z","iopub.status.idle":"2025-08-12T16:14:03.797943Z","shell.execute_reply.started":"2025-08-12T16:14:03.177856Z","shell.execute_reply":"2025-08-12T16:14:03.796542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate and print the percentage of each class to highlight imbalance\ntotal_rows = len(df_train)\nbinds_counts = df_train['binds'].value_counts()\npercentage_binds_1 = (binds_counts.get(1, 0) / total_rows) * 100\npercentage_binds_0 = (binds_counts.get(0, 0) / total_rows) * 100\n\nprint(f\"\\nPercentage of 'binds=1' (positive class): {percentage_binds_1:.2f}%\")\nprint(f\"Percentage of 'binds=0' (negative class): {percentage_binds_0:.2f}%\")\n\n# Interpretation\nprint(\"\\nInsight: The dataset is highly imbalanced, with only a small percentage of positive binding events. This is a common challenge in drug discovery datasets and will require careful consideration during model evaluation and potentially during training (e.g., using stratified sampling or class weights).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:14:03.799725Z","iopub.execute_input":"2025-08-12T16:14:03.801203Z","iopub.status.idle":"2025-08-12T16:14:03.811518Z","shell.execute_reply.started":"2025-08-12T16:14:03.80116Z","shell.execute_reply":"2025-08-12T16:14:03.809714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze the 'protein_name' feature\nprint(\"\\nUnique protein names and their counts:\")\nprotein_counts = df_train['protein_name'].value_counts()\nprint(protein_counts)\n\n# Visualize the distribution of protein names\nplt.figure(figsize=(10, 6))\nsns.countplot(y='protein_name', data=df_train, order=protein_counts.index)\nplt.title('Distribution of Protein Names')\nplt.xlabel('Count')\nplt.ylabel('Protein Name')\nplt.tight_layout()\nplt.show()\n\n# Insight\nprint(\"\\nInsight: The distribution of 'protein_name' is not uniform. Some proteins appear much more frequently than others. This suggests the data is not randomly sampled across all proteins and may impact how our model learns to generalize.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:14:03.812327Z","iopub.execute_input":"2025-08-12T16:14:03.812582Z","iopub.status.idle":"2025-08-12T16:14:04.092949Z","shell.execute_reply.started":"2025-08-12T16:14:03.812562Z","shell.execute_reply":"2025-08-12T16:14:04.090909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create new features for the length of SMILES strings\ndf_train['molecule_smiles_length'] = df_train['molecule_smiles'].apply(len)\ndf_train['buildingblock1_smiles_length'] = df_train['buildingblock1_smiles'].apply(len)\n\n# Analyze the distribution of these new features\nprint(\"\\nDescriptive statistics for molecule and building block SMILES lengths:\")\nprint(df_train[['molecule_smiles_length', 'buildingblock1_smiles_length']].describe())\n\n# Visualize the distribution of molecule SMILES length\nplt.figure(figsize=(10, 5))\nsns.histplot(df_train['molecule_smiles_length'], bins=50, kde=True)\nplt.title('Distribution of Molecule SMILES Lengths')\nplt.xlabel('Molecule SMILES Length')\nplt.ylabel('Frequency')\nplt.show()\n\n# Insight\nprint(\"\\nInsight: Most molecules have a SMILES string length within a certain range, but there are some outliers. This distribution can influence how we encode these features for our model.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:14:04.096009Z","iopub.execute_input":"2025-08-12T16:14:04.096398Z","iopub.status.idle":"2025-08-12T16:14:05.011617Z","shell.execute_reply.started":"2025-08-12T16:14:04.096373Z","shell.execute_reply":"2025-08-12T16:14:05.010538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compare molecule length for positive and negative binding events\nplt.figure(figsize=(10, 6))\nsns.boxplot(x='binds', y='molecule_smiles_length', data=df_train)\nplt.title('Molecule SMILES Length vs. Binding Status')\nplt.xlabel('Binds (0 = No, 1 = Yes)')\nplt.ylabel('Molecule SMILES Length')\nplt.show()\n\n# Insight\nprint(\"\\nInsight: The box plot shows whether there is a difference in the length of molecules that bind versus those that do not. A significant difference might indicate that molecule size is a useful feature for our model.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:14:05.012944Z","iopub.execute_input":"2025-08-12T16:14:05.013337Z","iopub.status.idle":"2025-08-12T16:14:05.237947Z","shell.execute_reply.started":"2025-08-12T16:14:05.013303Z","shell.execute_reply":"2025-08-12T16:14:05.236701Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Task 2: Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# Combine unique SMILES strings from train and test sets to fit the encoder\nall_smiles = pd.concat([df_train['molecule_smiles'], df_test['molecule_smiles'],\n                        df_train['buildingblock1_smiles'], df_test['buildingblock1_smiles'],\n                        df_train['buildingblock2_smiles'], df_test['buildingblock2_smiles'],\n                        df_train['buildingblock3_smiles'], df_test['buildingblock3_smiles']]).unique()\n\n# Initialize and fit the LabelEncoder on all unique SMILES strings\nfrom sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\nle.fit(all_smiles)\n\n# Preprocess training data\ndf_train_processed = pd.get_dummies(df_train, columns=['protein_name'], dtype=int)\ndf_train_processed['molecule_smiles_encoded'] = le.transform(df_train_processed['molecule_smiles'])\ndf_train_processed['buildingblock1_smiles_encoded'] = le.transform(df_train_processed['buildingblock1_smiles'])\ndf_train_processed['buildingblock2_smiles_encoded'] = le.transform(df_train_processed['buildingblock2_smiles'])\ndf_train_processed['buildingblock3_smiles_encoded'] = le.transform(df_train_processed['buildingblock3_smiles'])\ndf_train_processed = df_train_processed.drop(columns=['id', 'molecule_smiles', 'buildingblock1_smiles', 'buildingblock2_smiles', 'buildingblock3_smiles'])\n\n# Preprocess test data\ndf_test_processed = pd.get_dummies(df_test, columns=['protein_name'], dtype=int)\ndf_test_processed['molecule_smiles_encoded'] = le.transform(df_test_processed['molecule_smiles'])\ndf_test_processed['buildingblock1_smiles_encoded'] = le.transform(df_test_processed['buildingblock1_smiles'])\ndf_test_processed['buildingblock2_smiles_encoded'] = le.transform(df_test_processed['buildingblock2_smiles'])\ndf_test_processed['buildingblock3_smiles_encoded'] = le.transform(df_test_processed['buildingblock3_smiles'])\nX_test = df_test_processed.drop(columns=['id', 'molecule_smiles', 'buildingblock1_smiles', 'buildingblock2_smiles', 'buildingblock3_smiles'])\n\n# Align columns to handle any one-hot encoding differences\ntrain_columns = list(df_train_processed.drop('binds', axis=1).columns)\ntest_columns = list(X_test.columns)\n\nfor col in train_columns:\n    if col not in test_columns:\n        X_test[col] = 0\nX_test = X_test[train_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:14:05.239246Z","iopub.execute_input":"2025-08-12T16:14:05.239647Z","iopub.status.idle":"2025-08-12T16:14:19.230531Z","shell.execute_reply.started":"2025-08-12T16:14:05.239592Z","shell.execute_reply":"2025-08-12T16:14:19.229235Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Task 3: Model","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\n\n# 1. Define features (X) and target (y)\nX = df_train_processed.drop('binds', axis=1)\ny = df_train_processed['binds']\n\nprint(\"Features (X) shape:\", X.shape)\nprint(\"Target (y) shape:\", y.shape)\n\n# 2. Split the data into training and validation sets\n# Using a 80/20 split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\nprint(\"\\nTraining set shapes:\")\nprint(\"X_train:\", X_train.shape)\nprint(\"y_train:\", y_train.shape)\nprint(\"\\nValidation set shapes:\")\nprint(\"X_val:\", X_val.shape)\nprint(\"y_val:\", y_val.shape)\n\n# 3.Logistic Regression model\n# Initialize the model\n# Due to the class imbalance,`class_weight='balanced'`\nmodel = LogisticRegression(class_weight='balanced', solver='liblinear', random_state=42)\n\nprint(\"\\nModel selected: Logistic Regression with balanced class weights.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:14:19.231792Z","iopub.execute_input":"2025-08-12T16:14:19.232092Z","iopub.status.idle":"2025-08-12T16:14:19.287806Z","shell.execute_reply.started":"2025-08-12T16:14:19.232069Z","shell.execute_reply":"2025-08-12T16:14:19.286687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Task 4: Training and Evaluation","metadata":{}},{"cell_type":"code","source":"# 1. Training the model\nprint(\"Training the model...\")\nmodel.fit(X_train, y_train)\nprint(\"Model training complete.\")\n\n# 2. Predictions on the validation set\n# Predict the class labels\ny_pred = model.predict(X_val)\n\n# Predicting the probabilities for the positive class (binds=1)\ny_pred_proba = model.predict_proba(X_val)[:, 1]\n\n# 3. Evaluation of the model's performance\nfrom sklearn.metrics import classification_report, roc_auc_score\n\nprint(\"\\n Model Evaluation \")\nprint(\"Classification Report:\")\nprint(classification_report(y_val, y_pred))\n\n# Calculate and print the ROC AUC score\nroc_auc = roc_auc_score(y_val, y_pred_proba)\nprint(f\"ROC AUC Score: {roc_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:21:05.98988Z","iopub.execute_input":"2025-08-12T16:21:05.990244Z","iopub.status.idle":"2025-08-12T16:21:06.833541Z","shell.execute_reply.started":"2025-08-12T16:21:05.99022Z","shell.execute_reply":"2025-08-12T16:21:06.83245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Task 5: Inference","metadata":{}},{"cell_type":"code","source":"\n\n# 1. Load the test dataset\ntest_data_path = '/kaggle/input/leash-BELKA/test.csv'\ndf_test = pd.read_csv(test_data_path)\n\nprint(\"Test dataset loaded successfully!\")\nprint(\"Test data shape:\", df_test.shape)\n\n# 2. Preprocess the test data\ndf_test_processed = pd.get_dummies(df_test, columns=['protein_name'], dtype=int)\n\n# Using fitted LabelEncoder instances to transform the SMILES columns\ndf_test_processed['molecule_smiles_encoded'] = le.transform(df_test_processed['molecule_smiles'])\ndf_test_processed['buildingblock1_smiles_encoded'] = le.transform(df_test_processed['buildingblock1_smiles'])\ndf_test_processed['buildingblock2_smiles_encoded'] = le.transform(df_test_processed['buildingblock2_smiles'])\ndf_test_processed['buildingblock3_smiles_encoded'] = le.transform(df_test_processed['buildingblock3_smiles'])\n\n# Align the test columns with the training columns to handle any missing protein names in the test set\nX_test = df_test_processed.drop(columns=['id', 'molecule_smiles', 'buildingblock1_smiles', 'buildingblock2_smiles', 'buildingblock3_smiles'])\n\n# Handle potential missing columns from one-hot encoding if a protein in test set is not in train set\ntrain_columns = list(X_train.columns)\ntest_columns = list(X_test.columns)\n\nfor col in train_columns:\n    if col not in test_columns:\n        X_test[col] = 0\n        \nX_test = X_test[train_columns]\n\nprint(\"\\nPreprocessed test data shape:\", X_test.shape)\n\n# 3. Make predictions on the preprocessed test data\ntest_predictions = model.predict(X_test)\nprint(\"\\nPredictions generated successfully.\")\n\n# 4. Create the submission file\nsubmission_df = pd.DataFrame({'id': df_test['id'], 'binds': test_predictions})\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"\\nSubmission file 'submission.csv' created successfully.\")\nprint(\"First 5 rows of the submission file:\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-12T16:14:20.159916Z","iopub.execute_input":"2025-08-12T16:14:20.160215Z","iopub.status.idle":"2025-08-12T16:14:31.655542Z","shell.execute_reply.started":"2025-08-12T16:14:20.160186Z","shell.execute_reply":"2025-08-12T16:14:31.654148Z"}},"outputs":[],"execution_count":null}]}