{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-18T12:03:28.257902Z","iopub.execute_input":"2024-05-18T12:03:28.258296Z","iopub.status.idle":"2024-05-18T12:03:29.355471Z","shell.execute_reply.started":"2024-05-18T12:03:28.258266Z","shell.execute_reply":"2024-05-18T12:03:29.354457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/leash-BELKA/train.csv',nrows = 1)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:31.817532Z","iopub.execute_input":"2024-05-18T12:03:31.818046Z","iopub.status.idle":"2024-05-18T12:03:31.841851Z","shell.execute_reply.started":"2024-05-18T12:03:31.818014Z","shell.execute_reply":"2024-05-18T12:03:31.840773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:34.639404Z","iopub.execute_input":"2024-05-18T12:03:34.640393Z","iopub.status.idle":"2024-05-18T12:03:34.659046Z","shell.execute_reply.started":"2024-05-18T12:03:34.640354Z","shell.execute_reply":"2024-05-18T12:03:34.657875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_df.columns:\n    print(i)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:39.026462Z","iopub.execute_input":"2024-05-18T12:03:39.026891Z","iopub.status.idle":"2024-05-18T12:03:39.032080Z","shell.execute_reply.started":"2024-05-18T12:03:39.026850Z","shell.execute_reply":"2024-05-18T12:03:39.030895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = 0 \nnew_train_df = pd.DataFrame()\n\nfor chunk in pd.read_csv('/kaggle/input/leash-BELKA/train.csv', chunksize=2000):\n    # Rename columns\n    chunk.columns = ['id','buildingblock1_smiles','buildingblock2_smiles','buildingblock3_smiles','molecule_smiles','protein_name','binds']\n    \n    # Append chunk to new_train_df\n    new_train_df = pd.concat([new_train_df, chunk], ignore_index=True)\n    \n    counter += 1\n    if counter == 25:\n        break\n        \nprint(new_train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:40.033154Z","iopub.execute_input":"2024-05-18T12:03:40.034097Z","iopub.status.idle":"2024-05-18T12:03:40.341107Z","shell.execute_reply.started":"2024-05-18T12:03:40.034060Z","shell.execute_reply":"2024-05-18T12:03:40.339907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:41.649080Z","iopub.execute_input":"2024-05-18T12:03:41.649502Z","iopub.status.idle":"2024-05-18T12:03:41.662663Z","shell.execute_reply.started":"2024-05-18T12:03:41.649464Z","shell.execute_reply":"2024-05-18T12:03:41.661600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:44.995939Z","iopub.execute_input":"2024-05-18T12:03:44.996334Z","iopub.status.idle":"2024-05-18T12:03:45.036569Z","shell.execute_reply.started":"2024-05-18T12:03:44.996301Z","shell.execute_reply":"2024-05-18T12:03:45.035527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:47.312662Z","iopub.execute_input":"2024-05-18T12:03:47.313059Z","iopub.status.idle":"2024-05-18T12:03:47.335340Z","shell.execute_reply.started":"2024-05-18T12:03:47.313027Z","shell.execute_reply":"2024-05-18T12:03:47.334192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_df['protein_name'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:49.181325Z","iopub.execute_input":"2024-05-18T12:03:49.182471Z","iopub.status.idle":"2024-05-18T12:03:49.195053Z","shell.execute_reply.started":"2024-05-18T12:03:49.182408Z","shell.execute_reply":"2024-05-18T12:03:49.193971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_df['binds'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:53.017411Z","iopub.execute_input":"2024-05-18T12:03:53.017828Z","iopub.status.idle":"2024-05-18T12:03:53.028042Z","shell.execute_reply.started":"2024-05-18T12:03:53.017800Z","shell.execute_reply":"2024-05-18T12:03:53.026931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv('/kaggle/input/leash-BELKA/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:53.542507Z","iopub.execute_input":"2024-05-18T12:03:53.542926Z","iopub.status.idle":"2024-05-18T12:03:59.924487Z","shell.execute_reply.started":"2024-05-18T12:03:53.542890Z","shell.execute_reply":"2024-05-18T12:03:59.923345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:03:59.926117Z","iopub.execute_input":"2024-05-18T12:03:59.926491Z","iopub.status.idle":"2024-05-18T12:03:59.934169Z","shell.execute_reply.started":"2024-05-18T12:03:59.926452Z","shell.execute_reply":"2024-05-18T12:03:59.933088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_data.sample(10000)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:04:04.288953Z","iopub.execute_input":"2024-05-18T12:04:04.289305Z","iopub.status.idle":"2024-05-18T12:04:04.298073Z","shell.execute_reply.started":"2024-05-18T12:04:04.289269Z","shell.execute_reply":"2024-05-18T12:04:04.296920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:04:06.932766Z","iopub.execute_input":"2024-05-18T12:04:06.933192Z","iopub.status.idle":"2024-05-18T12:04:06.939809Z","shell.execute_reply.started":"2024-05-18T12:04:06.933157Z","shell.execute_reply":"2024-05-18T12:04:06.938735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:04:10.053716Z","iopub.execute_input":"2024-05-18T12:04:10.054102Z","iopub.status.idle":"2024-05-18T12:04:25.727869Z","shell.execute_reply.started":"2024-05-18T12:04:10.054072Z","shell.execute_reply":"2024-05-18T12:04:25.726505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem\n\n# Define a function to convert SMILES to fingerprints\ndef smiles_to_fingerprint(smiles):\n    # Convert SMILES to RDKit molecule object\n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return None\n    # Generate Morgan fingerprint for the molecule\n    fingerprint = AllChem.GetMorganFingerprintAsBitVect(mol, 2)  # Radius 2\n    return fingerprint","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:04:27.770548Z","iopub.execute_input":"2024-05-18T12:04:27.771441Z","iopub.status.idle":"2024-05-18T12:04:27.996571Z","shell.execute_reply.started":"2024-05-18T12:04:27.771380Z","shell.execute_reply":"2024-05-18T12:04:27.995558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the function to the 'molecule_smiles' column\nnew_train_df['molecule_fingerprint'] = new_train_df['molecule_smiles'].apply(smiles_to_fingerprint)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:04:32.032079Z","iopub.execute_input":"2024-05-18T12:04:32.032459Z","iopub.status.idle":"2024-05-18T12:04:53.706521Z","shell.execute_reply.started":"2024-05-18T12:04:32.032415Z","shell.execute_reply":"2024-05-18T12:04:53.705566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the function to the 'buildingblock1_smiles' column\nnew_train_df['buildingblock1_fingerprint'] = new_train_df['buildingblock1_smiles'].apply(smiles_to_fingerprint)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:04:53.708290Z","iopub.execute_input":"2024-05-18T12:04:53.708644Z","iopub.status.idle":"2024-05-18T12:05:08.309253Z","shell.execute_reply.started":"2024-05-18T12:04:53.708614Z","shell.execute_reply":"2024-05-18T12:05:08.308243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the function to the 'buildingblock2_smiles' column\nnew_train_df['buildingblock2_fingerprint'] = new_train_df['buildingblock2_smiles'].apply(smiles_to_fingerprint)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:05:08.310561Z","iopub.execute_input":"2024-05-18T12:05:08.310954Z","iopub.status.idle":"2024-05-18T12:05:13.443685Z","shell.execute_reply.started":"2024-05-18T12:05:08.310924Z","shell.execute_reply":"2024-05-18T12:05:13.442268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the function to the 'buildingblock3_smiles' column\nnew_train_df['buildingblock3_fingerprint'] = new_train_df['buildingblock3_smiles'].apply(smiles_to_fingerprint)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:05:13.446002Z","iopub.execute_input":"2024-05-18T12:05:13.446443Z","iopub.status.idle":"2024-05-18T12:05:19.446689Z","shell.execute_reply.started":"2024-05-18T12:05:13.446386Z","shell.execute_reply":"2024-05-18T12:05:19.445632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform one-hot encoding on the 'protein_name' column\nencoded_protein_names = pd.get_dummies(new_train_df['protein_name'], prefix='protein', drop_first=True)\n# Convert True and False to 1s and 0s\nencoded_protein_names = encoded_protein_names.astype(int)\n# Concatenate the encoded columns with the original DataFrame\nnew_train_df = pd.concat([new_train_df, encoded_protein_names], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:06:11.347678Z","iopub.execute_input":"2024-05-18T12:06:11.348474Z","iopub.status.idle":"2024-05-18T12:06:11.395371Z","shell.execute_reply.started":"2024-05-18T12:06:11.348435Z","shell.execute_reply":"2024-05-18T12:06:11.394249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:06:15.370714Z","iopub.execute_input":"2024-05-18T12:06:15.371118Z","iopub.status.idle":"2024-05-18T12:06:15.410337Z","shell.execute_reply.started":"2024-05-18T12:06:15.371085Z","shell.execute_reply":"2024-05-18T12:06:15.409252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop the original 'protein_name' column\nnew_train_df.drop(columns=['id','buildingblock1_smiles','buildingblock2_smiles','buildingblock3_smiles','molecule_smiles','protein_name'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:07:47.994190Z","iopub.execute_input":"2024-05-18T12:07:47.994631Z","iopub.status.idle":"2024-05-18T12:07:48.008146Z","shell.execute_reply.started":"2024-05-18T12:07:47.994597Z","shell.execute_reply":"2024-05-18T12:07:48.006988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_df","metadata":{"execution":{"iopub.status.busy":"2024-05-18T12:08:00.232278Z","iopub.execute_input":"2024-05-18T12:08:00.232756Z","iopub.status.idle":"2024-05-18T12:08:00.290676Z","shell.execute_reply.started":"2024-05-18T12:08:00.232719Z","shell.execute_reply":"2024-05-18T12:08:00.289486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train_df.to_csv('mysubmission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink('mysubmission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}