{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:35:49.908362Z","iopub.execute_input":"2024-07-05T07:35:49.908670Z","iopub.status.idle":"2024-07-05T07:36:05.447317Z","shell.execute_reply.started":"2024-07-05T07:35:49.908642Z","shell.execute_reply":"2024-07-05T07:36:05.446079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install duckdb","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:36:05.449711Z","iopub.execute_input":"2024-07-05T07:36:05.450104Z","iopub.status.idle":"2024-07-05T07:36:18.694498Z","shell.execute_reply.started":"2024-07-05T07:36:05.450067Z","shell.execute_reply":"2024-07-05T07:36:18.693375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport rdkit\nimport duckdb\n\nfrom rdkit import Chem\nfrom rdkit.Chem import rdFingerprintGenerator\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import train_test_split\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:36:18.696054Z","iopub.execute_input":"2024-07-05T07:36:18.696382Z","iopub.status.idle":"2024-07-05T07:36:20.143798Z","shell.execute_reply.started":"2024-07-05T07:36:18.696350Z","shell.execute_reply":"2024-07-05T07:36:20.143024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path ='/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:36:20.145756Z","iopub.execute_input":"2024-07-05T07:36:20.146161Z","iopub.status.idle":"2024-07-05T07:36:20.150668Z","shell.execute_reply.started":"2024-07-05T07:36:20.146135Z","shell.execute_reply":"2024-07-05T07:36:20.149559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con = duckdb.connect()\ndf=con.query(f\"\"\"(SELECT * \n                FROM parquet_scan('{train_path}')\n                WHERE binds = 0\n                ORDER BY random()\n                LIMIT 50000)\n                UNION ALL\n                (SELECT *\n                FROM parquet_scan('{train_path}')\n                WHERE BINDS = 1\n                ORDER BY random()\n                LIMIT 50000)\"\"\").df()\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:39:23.320415Z","iopub.execute_input":"2024-07-05T07:39:23.320789Z","iopub.status.idle":"2024-07-05T07:40:09.447677Z","shell.execute_reply.started":"2024-07-05T07:39:23.320758Z","shell.execute_reply":"2024-07-05T07:40:09.446859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:38:27.466739Z","iopub.execute_input":"2024-07-05T07:38:27.467056Z","iopub.status.idle":"2024-07-05T07:38:27.471730Z","shell.execute_reply.started":"2024-07-05T07:38:27.467032Z","shell.execute_reply":"2024-07-05T07:38:27.470819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:40:09.451216Z","iopub.execute_input":"2024-07-05T07:40:09.451499Z","iopub.status.idle":"2024-07-05T07:40:37.394855Z","shell.execute_reply.started":"2024-07-05T07:40:09.451474Z","shell.execute_reply":"2024-07-05T07:40:37.394021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:40:37.395962Z","iopub.execute_input":"2024-07-05T07:40:37.396237Z","iopub.status.idle":"2024-07-05T07:40:37.410036Z","shell.execute_reply.started":"2024-07-05T07:40:37.396212Z","shell.execute_reply":"2024-07-05T07:40:37.409222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mfpgen = rdFingerprintGenerator.GetMorganGenerator(radius=2,fpSize=2048)\n\ndf['ecfp'] = df['molecule'].apply(mfpgen.GetFingerprint).apply(lambda bitvec: list(bitvec))","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:41:20.333438Z","iopub.execute_input":"2024-07-05T07:41:20.333909Z","iopub.status.idle":"2024-07-05T07:44:27.629124Z","shell.execute_reply.started":"2024-07-05T07:41:20.333876Z","shell.execute_reply":"2024-07-05T07:44:27.628322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"onehot_encoder =OneHotEncoder(sparse_output=False)\nprotein_onehot =onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:44:27.630961Z","iopub.execute_input":"2024-07-05T07:44:27.631393Z","iopub.status.idle":"2024-07-05T07:44:27.674722Z","shell.execute_reply.started":"2024-07-05T07:44:27.631359Z","shell.execute_reply":"2024-07-05T07:44:27.673825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = [ecfp + protein for ecfp,protein in zip(df['ecfp'].tolist(),protein_onehot.tolist())]\ny = df['binds'].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:44:27.675797Z","iopub.execute_input":"2024-07-05T07:44:27.676102Z","iopub.status.idle":"2024-07-05T07:44:32.904673Z","shell.execute_reply.started":"2024-07-05T07:44:27.676078Z","shell.execute_reply":"2024-07-05T07:44:32.903847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#rf_model = RandomForestClassifier(n_estimators=100, random_state=42)\n#rf_model.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:37:31.304213Z","iopub.status.idle":"2024-07-05T07:37:31.304542Z","shell.execute_reply.started":"2024-07-05T07:37:31.304384Z","shell.execute_reply":"2024-07-05T07:37:31.304398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on the test set\n#y_pred_proba = rf_model.predict_proba(X_test)[:, 1]  # Probability of the positive class\n\n# Calculate the mean average precision\n#map_score = average_precision_score(Y_test, y_pred_proba)\n#print(f\"Mean Average Precision (mAP): {map_score:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:37:31.306367Z","iopub.status.idle":"2024-07-05T07:37:31.306800Z","shell.execute_reply.started":"2024-07-05T07:37:31.306580Z","shell.execute_reply":"2024-07-05T07:37:31.306598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(X[0])","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:37:31.308133Z","iopub.status.idle":"2024-07-05T07:37:31.308456Z","shell.execute_reply.started":"2024-07-05T07:37:31.308298Z","shell.execute_reply":"2024-07-05T07:37:31.308313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_test,Y_train,Y_test = train_test_split(X,y,test_size=0.2,random_state=407)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:50:25.939902Z","iopub.execute_input":"2024-07-05T07:50:25.940273Z","iopub.status.idle":"2024-07-05T07:50:25.998463Z","shell.execute_reply.started":"2024-07-05T07:50:25.940244Z","shell.execute_reply":"2024-07-05T07:50:25.997492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import xgboost as xgb\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:50:35.902756Z","iopub.execute_input":"2024-07-05T07:50:35.903516Z","iopub.status.idle":"2024-07-05T07:50:38.526411Z","shell.execute_reply.started":"2024-07-05T07:50:35.903482Z","shell.execute_reply":"2024-07-05T07:50:38.525562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_train = lgb.Dataset(np.array(X_train),label=Y_train)\nlgb_valid = lgb.Dataset(np.array(X_test),label=Y_test,reference= lgb_train)\n\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 3,\n    \"num_leaves\": 31,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"n_estimators\": 1000,\n    \"verbose\": -1,\n}\n\ngbm = lgb.train(\n    params,\n    lgb_train,\n    valid_sets=lgb_valid,\n    callbacks=[lgb.log_evaluation(50), lgb.early_stopping(10)]\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:50:38.528193Z","iopub.execute_input":"2024-07-05T07:50:38.529204Z","iopub.status.idle":"2024-07-05T07:51:47.784426Z","shell.execute_reply.started":"2024-07-05T07:50:38.529168Z","shell.execute_reply":"2024-07-05T07:51:47.783516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on the test set\ny_pred_proba = gbm.predict(X_test)  # Probability of the positive class\n\n# Calculate the mean average precision\nmap_score = average_precision_score(Y_test, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-05T07:51:52.212786Z","iopub.execute_input":"2024-07-05T07:51:52.213270Z","iopub.status.idle":"2024-07-05T07:51:57.941312Z","shell.execute_reply.started":"2024-07-05T07:51:52.213236Z","shell.execute_reply":"2024-07-05T07:51:57.940268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os","metadata":{"execution":{"iopub.status.busy":"2024-07-05T08:16:20.047094Z","iopub.execute_input":"2024-07-05T08:16:20.047868Z","iopub.status.idle":"2024-07-05T08:16:20.052012Z","shell.execute_reply.started":"2024-07-05T08:16:20.047824Z","shell.execute_reply":"2024-07-05T08:16:20.050843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_file='/kaggle/input/leash-BELKA/test.csv'\noutput_file='submission.csv'\n#df_test= pd.read_csv(test_file)\nn=0\nfor df_test in pd.read_csv(test_file,chunksize=100000):\n    n+=len(df_test)\n    print(f'number of test set: {n}')\n    df_test['molecule'] = df_test['molecule_smiles'].apply(Chem.MolFromSmiles)\n    \n    mfpgen = rdFingerprintGenerator.GetMorganGenerator(radius=2,fpSize=2048)\n    df_test['ecfp'] = df_test['molecule'].apply(mfpgen.GetFingerprint).apply(lambda bitvec: list(bitvec))\n    protein_one_hot = onehot_encoder.transform(df_test['protein_name'].values.reshape(-1,1))\n\n    X_test = [ecfp + protein for ecfp, protein in zip(df_test['ecfp'].tolist(), protein_onehot.tolist())]\n    \n    # Predict the probabilities\n    probabilities = gbm.predict(X_test)\n\n    # Create a DataFrame with 'id' and 'probability' columns\n    output_df = pd.DataFrame({'id': df_test['id'], 'binds': probabilities})\n\n    # Save the output DataFrame to a CSV file\n    output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))","metadata":{"execution":{"iopub.status.busy":"2024-07-05T08:18:50.000488Z","iopub.execute_input":"2024-07-05T08:18:50.001321Z","iopub.status.idle":"2024-07-05T09:28:20.248296Z","shell.execute_reply.started":"2024-07-05T08:18:50.001290Z","shell.execute_reply":"2024-07-05T09:28:20.247313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.read_csv('/kaggle/working/submission.csv')\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-05T09:33:22.522268Z","iopub.execute_input":"2024-07-05T09:33:22.523013Z","iopub.status.idle":"2024-07-05T09:33:23.049583Z","shell.execute_reply.started":"2024-07-05T09:33:22.522981Z","shell.execute_reply":"2024-07-05T09:33:23.048472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}