{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install rdkit","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-10T08:58:06.216330Z","iopub.execute_input":"2024-06-10T08:58:06.216749Z","iopub.status.idle":"2024-06-10T08:58:25.426621Z","shell.execute_reply.started":"2024-06-10T08:58:06.216716Z","shell.execute_reply":"2024-06-10T08:58:25.425159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install duckdb\n","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:58:25.429001Z","iopub.execute_input":"2024-06-10T08:58:25.429932Z","iopub.status.idle":"2024-06-10T08:58:41.541329Z","shell.execute_reply.started":"2024-06-10T08:58:25.429895Z","shell.execute_reply":"2024-06-10T08:58:41.540090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\n\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 30000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 30000)\"\"\").df()\n\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:58:41.543072Z","iopub.execute_input":"2024-06-10T08:58:41.543480Z","iopub.status.idle":"2024-06-10T08:59:40.881760Z","shell.execute_reply.started":"2024-06-10T08:58:41.543444Z","shell.execute_reply":"2024-06-10T08:59:40.880529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:59:40.884988Z","iopub.execute_input":"2024-06-10T08:59:40.885944Z","iopub.status.idle":"2024-06-10T08:59:40.912430Z","shell.execute_reply.started":"2024-06-10T08:59:40.885906Z","shell.execute_reply":"2024-06-10T08:59:40.911170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nplt.figure(figsize = (10, 8))\nsns.countplot(data = df, x = \"protein_name\", hue = \"binds\", palette = \"pastel\", dodge = True)\nplt.title(\"protein influence in Binding\")\nplt.xlabel(\"protein\")\nplt.ylabel(\"Binding\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:59:40.913809Z","iopub.execute_input":"2024-06-10T08:59:40.914195Z","iopub.status.idle":"2024-06-10T08:59:42.872897Z","shell.execute_reply.started":"2024-06-10T08:59:40.914157Z","shell.execute_reply":"2024-06-10T08:59:42.871582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_percent = df.groupby([\"protein_name\", \"binds\"]).size().groupby(level = 0).apply(lambda x : 100 * x / float(x.sum())).unstack().fillna(0) \nprotein_percent.plot(kind='bar', stacked=True, color=['green', 'red'], figsize=(10, 6))\nplt.title(\"protein influence in Binding\")\nplt.xlabel(\"protein\")\nplt.ylabel(\"Binding\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:59:42.874439Z","iopub.execute_input":"2024-06-10T08:59:42.874788Z","iopub.status.idle":"2024-06-10T08:59:43.187780Z","shell.execute_reply.started":"2024-06-10T08:59:42.874758Z","shell.execute_reply":"2024-06-10T08:59:43.186535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_percent","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:59:43.189565Z","iopub.execute_input":"2024-06-10T08:59:43.190036Z","iopub.status.idle":"2024-06-10T08:59:43.204623Z","shell.execute_reply.started":"2024-06-10T08:59:43.189994Z","shell.execute_reply":"2024-06-10T08:59:43.203222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_percent.plot(kind='bar', stacked=True, color=['green', 'red'], figsize=(10, 6))","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:59:43.206102Z","iopub.execute_input":"2024-06-10T08:59:43.206542Z","iopub.status.idle":"2024-06-10T08:59:43.504047Z","shell.execute_reply.started":"2024-06-10T08:59:43.206505Z","shell.execute_reply":"2024-06-10T08:59:43.502806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem \nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:59:43.505710Z","iopub.execute_input":"2024-06-10T08:59:43.506153Z","iopub.status.idle":"2024-06-10T08:59:44.391758Z","shell.execute_reply.started":"2024-06-10T08:59:43.506096Z","shell.execute_reply":"2024-06-10T08:59:44.390521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-10T08:59:44.395752Z","iopub.execute_input":"2024-06-10T08:59:44.396172Z","iopub.status.idle":"2024-06-10T09:00:04.818960Z","shell.execute_reply.started":"2024-06-10T08:59:44.396118Z","shell.execute_reply":"2024-06-10T09:00:04.817792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['molecule']","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:00:04.820247Z","iopub.execute_input":"2024-06-10T09:00:04.820591Z","iopub.status.idle":"2024-06-10T09:00:04.834902Z","shell.execute_reply.started":"2024-06-10T09:00:04.820563Z","shell.execute_reply":"2024-06-10T09:00:04.833560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_ecfp(molecule, radius = 2, bits = 1024):\n    if molecule is None:\n        return None\n    return list(AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:00:04.836582Z","iopub.execute_input":"2024-06-10T09:00:04.837044Z","iopub.status.idle":"2024-06-10T09:00:04.849976Z","shell.execute_reply.started":"2024-06-10T09:00:04.836998Z","shell.execute_reply":"2024-06-10T09:00:04.848796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"ecfp\"] = df[\"molecule\"].apply(generate_ecfp)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:00:04.851691Z","iopub.execute_input":"2024-06-10T09:00:04.852497Z","iopub.status.idle":"2024-06-10T09:01:31.986742Z","shell.execute_reply.started":"2024-06-10T09:00:04.852456Z","shell.execute_reply":"2024-06-10T09:01:31.985547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"ecfp\"]","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:01:31.988121Z","iopub.execute_input":"2024-06-10T09:01:31.988544Z","iopub.status.idle":"2024-06-10T09:01:32.005069Z","shell.execute_reply.started":"2024-06-10T09:01:31.988506Z","shell.execute_reply":"2024-06-10T09:01:32.003891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"onehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_onehot = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:01:32.006697Z","iopub.execute_input":"2024-06-10T09:01:32.007060Z","iopub.status.idle":"2024-06-10T09:01:32.050540Z","shell.execute_reply.started":"2024-06-10T09:01:32.007030Z","shell.execute_reply":"2024-06-10T09:01:32.049536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = [ecfp + protein for ecfp, protein in zip(df['ecfp'].tolist(), protein_onehot.tolist())]\ny = df['binds'].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:01:32.052145Z","iopub.execute_input":"2024-06-10T09:01:32.052519Z","iopub.status.idle":"2024-06-10T09:01:33.820852Z","shell.execute_reply.started":"2024-06-10T09:01:32.052488Z","shell.execute_reply":"2024-06-10T09:01:33.819712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X[0])","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:01:33.822344Z","iopub.execute_input":"2024-06-10T09:01:33.822778Z","iopub.status.idle":"2024-06-10T09:01:33.833787Z","shell.execute_reply.started":"2024-06-10T09:01:33.822738Z","shell.execute_reply":"2024-06-10T09:01:33.832036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\npca = PCA(n_components = 0.95)\nX_pca = pca.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:01:33.835483Z","iopub.execute_input":"2024-06-10T09:01:33.835956Z","iopub.status.idle":"2024-06-10T09:02:07.745113Z","shell.execute_reply.started":"2024-06-10T09:01:33.835912Z","shell.execute_reply":"2024-06-10T09:02:07.742683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\ncumulative_variance = np.cumsum(pca.explained_variance_ratio_)\nplt.figure(figsize = (8, 6))\nplt.plot(range(1, len(cumulative_variance) + 1), cumulative_variance, marker='o', linestyle='--')\nplt.xlabel('Number of Dimensions')\nplt.ylabel('Cumulative Explained Variance Ratio')\nplt.title('Cumulative Explained Variance Ratio by Number of Dimensions')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T07:35:40.872361Z","iopub.execute_input":"2024-06-09T07:35:40.872731Z","iopub.status.idle":"2024-06-09T07:35:41.165993Z","shell.execute_reply.started":"2024-06-09T07:35:40.872698Z","shell.execute_reply":"2024-06-09T07:35:41.164727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import KMeans \nfrom sklearn.metrics import silhouette_score\n\nsilhouette_scores = []\n\nfor cluster in range(2, 9):\n    kmeans = KMeans(n_clusters = cluster, init='k-means++', n_init = 10, max_iter = 100, random_state = 0)\n    kmeans.fit(X_pca)\n    silhouette_scores.append(silhouette_score(X_pca, kmeans.labels_))\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-06-09T07:35:41.167484Z","iopub.execute_input":"2024-06-09T07:35:41.168297Z","iopub.status.idle":"2024-06-09T07:47:10.316364Z","shell.execute_reply.started":"2024-06-09T07:35:41.168254Z","shell.execute_reply":"2024-06-09T07:47:10.313845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"silhouette_scores","metadata":{"execution":{"iopub.status.busy":"2024-06-09T07:47:10.319440Z","iopub.execute_input":"2024-06-09T07:47:10.319992Z","iopub.status.idle":"2024-06-09T07:47:10.332054Z","shell.execute_reply.started":"2024-06-09T07:47:10.319945Z","shell.execute_reply":"2024-06-09T07:47:10.330626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10, 8))\nplt.plot(range(2,9), silhouette_scores, marker = 'o', linestyle = '--')\nplt.xlabel('Number of Clusters')\nplt.ylabel('Silhouette Score')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T07:47:10.333985Z","iopub.execute_input":"2024-06-09T07:47:10.334441Z","iopub.status.idle":"2024-06-09T07:47:10.644318Z","shell.execute_reply.started":"2024-06-09T07:47:10.334401Z","shell.execute_reply":"2024-06-09T07:47:10.642747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_test, y_train, y_test = train_test_split(X_pca, y, test_size = 0.2, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:10:05.079700Z","iopub.execute_input":"2024-06-10T09:10:05.080266Z","iopub.status.idle":"2024-06-10T09:10:05.503617Z","shell.execute_reply.started":"2024-06-10T09:10:05.080227Z","shell.execute_reply":"2024-06-10T09:10:05.502178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier \nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import roc_curve, auc, confusion_matrix, accuracy_score, classification_report\nfrom sklearn.model_selection import cross_val_score\n\nmodels = {\n    \n    \"Randome forest\" : RandomForestClassifier(),\n    \"XGBoost\" : XGBClassifier(\n        n_estimators = 100,\n        max_depth = 150,\n        learning_rate = 0.1,\n        subsample = 0.8,\n        colsample_bytree = 0.8,\n        objective = \"multi:softmax\",\n        num_class = 2\n        \n    ),\n    \"LGBM\" : LGBMClassifier(\n        boosting_type = 'gbdt',\n        bagging_freq = 5,\n        verbose = 0,\n        num_leaves = 31,\n        max_depth = 250,\n        learning_rate = 0.1,\n        n_estimators = 100\n    )\n}\n\nfor name, model in tqdm(models.items(), desc = \"Training models\", total = len(models)):\n    model.fit(x_train, y_train)\n    score_training = cross_val_score(model, x_train, y_train, cv = 10)\n    \n    pred = model.predict(x_test)\n    tqdm.write(\"Model: {} has Accuracy {:.2f}%\".format(model.__class__.__name__, round(score_training.mean(), 2) * 100))\n    print()\n    ","metadata":{"execution":{"iopub.status.busy":"2024-06-10T09:18:03.532378Z","iopub.execute_input":"2024-06-10T09:18:03.532808Z","iopub.status.idle":"2024-06-10T11:23:18.547847Z","shell.execute_reply.started":"2024-06-10T09:18:03.532773Z","shell.execute_reply":"2024-06-10T11:23:18.545784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for name, model in models.items():\n    y_pred = model.predict(x_test)\n    \n    fpr, tpr, _ = roc_curve(y_test, model.predict_proba(x_test)[:, 1])\n    roc_auc = auc(fpr, tpr)\n    \n    print()\n    \n    plt.figure()\n    plt.plot(fpr, tpr, color = 'darkorange', lw = 2, label = 'ROC curve (area = %0.2f)' % roc_auc)\n    plt.plot([0, 1], [0, 1], color = 'navy', lw = 2, linestyle = '--')\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver Operating Characteristic - {}'.format(name))\n    plt.grid(False)\n    plt.show()\n    \n    acc = accuracy_score(y_test, y_pred)\n    print(\"Accuracy : \", acc)\n    print()\n    \n    cm = confusion_matrix(y_test, y_pred)\n    print()\n    \n    plt.figure(figsize = (8, 6))\n    sns.heatmap(cm, annot = True, fmt = 'd', cmap = 'Blues')\n    plt.title('Confusion Matrix - {}'.format(name))\n    plt.show()\n    \n    print(\"Classification report\")\n    print(classification_report(y_test, y_pred))\n    ","metadata":{"execution":{"iopub.status.busy":"2024-06-10T12:04:09.956017Z","iopub.execute_input":"2024-06-10T12:04:09.956531Z","iopub.status.idle":"2024-06-10T12:04:17.087495Z","shell.execute_reply.started":"2024-06-10T12:04:09.956458Z","shell.execute_reply":"2024-06-10T12:04:17.085971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}