{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is a code-only fork of [Andrew Blevin's Leash-BELKA tutorial code](https://www.kaggle.com/code/andrewdblevins/leash-tutorial-ecfps-and-random-forest)\n\nIt keeps the same overall logic as the original, but switching to numpy arrays and\nrearranging the code a bit lets this version run comfortably on an 8Gb laptop.\n\nStart-to-finish, it takes 20 minutes in a Kaggle notebook.\n\n(The original runs out of memory, even with a lot of swap space, on small machines.)\n\nHave fun.","metadata":{}},{"cell_type":"code","source":"# RDkit, duckDB\n\n!pip install rdkit\n\n!pip install duckdb","metadata":{"execution":{"iopub.status.busy":"2024-04-10T18:15:17.827824Z","iopub.execute_input":"2024-04-10T18:15:17.828267Z","iopub.status.idle":"2024-04-10T18:15:47.826641Z","shell.execute_reply.started":"2024-04-10T18:15:17.828235Z","shell.execute_reply":"2024-04-10T18:15:47.825172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#%%\n# imports\n\nimport numpy as np\nimport pandas as pd\nimport os\n\nimport duckdb\n\nfrom rdkit import Chem\nfrom rdkit.Chem import rdFingerprintGenerator\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\n\n\n# %%\n# fetch train data subset\n\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\n# creates a balanced set\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 30000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 30000)\"\"\").df()\n\ncon.close()\n\n\n#%%\n# transform training data\n\nmfpgen = rdFingerprintGenerator.GetMorganGenerator(radius=2,fpSize=1024)\n\ndef ecfp_from_smiles(smiles):\n    if smiles is None:\n        return None\n    molecule = Chem.MolFromSmiles(smiles)\n    if molecule is None:\n        return None\n    return mfpgen.GetFingerprintAsNumPy(molecule)\n\nX = df['molecule_smiles'].apply(ecfp_from_smiles)\n\n# de-nest the array\nX = np.array(list(X))\n\n# one-hot encode the protein_name\nonehot_encoder = OneHotEncoder(categories=[['BRD4', 'HSA', 'sEH']], sparse_output=False)\nX = np.hstack([X, onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))])\n\ny = df['binds'].values.reshape(-1)\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n\n#%%\n# fit a model\n\n# Create and train the random forest model\nrf_model = RandomForestClassifier(n_estimators=100, max_depth=32, random_state=42)\nrf_model.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred_proba = rf_model.predict_proba(X_test)[:, 1]  # Probability of the positive class\n\n# Calculate the mean average precision\nmap_score = average_precision_score(y_test, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.2f}\")\nprint(f'{np.sum(y_test)} positive of {len(y_test)}')\n\n\n#%%\n# write submission file\n\n# Process the test file chunk by chunk\ntest_file = '/kaggle/input/leash-BELKA/test.csv'\noutput_file = 'submission.csv'  # Specify the path and filename for the output file\n\nn = 0\n# Read a chunk into a pandas DataFrame\nfor df_test in pd.read_csv(test_file, chunksize=100000):\n\n    X_test = df_test['molecule_smiles'].apply(ecfp_from_smiles)\n\n    # de-nest the array\n    X_test = np.array(list(X_test))\n\n    # one-hot encode the protein_name\n    X_test = np.hstack([X_test, onehot_encoder.fit_transform(df_test['protein_name'].values.reshape(-1, 1))])\n\n    # # Predict the probabilities\n    probabilities = rf_model.predict_proba(X_test)[:, 1]\n\n    # Create a DataFrame with 'id' and 'probability' columns\n    output_df = pd.DataFrame({'id': df_test['id'], 'binds': probabilities})\n\n    # Save the output DataFrame to a CSV file\n    output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))\n    \n    n += len(df_test)\n    print(f'processed {n}')\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-10T18:20:48.915822Z","iopub.execute_input":"2024-04-10T18:20:48.916486Z","iopub.status.idle":"2024-04-10T18:40:27.297968Z","shell.execute_reply.started":"2024-04-10T18:20:48.916451Z","shell.execute_reply":"2024-04-10T18:40:27.296453Z"},"trusted":true},"execution_count":null,"outputs":[]}]}