{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../inpdirectory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-10T11:04:37.303474Z","iopub.execute_input":"2024-04-10T11:04:37.303974Z","iopub.status.idle":"2024-04-10T11:04:37.315548Z","shell.execute_reply.started":"2024-04-10T11:04:37.303937Z","shell.execute_reply":"2024-04-10T11:04:37.314307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install deepchem[tensorflow]","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:30:41.161928Z","iopub.execute_input":"2024-04-10T12:30:41.162653Z","iopub.status.idle":"2024-04-10T12:30:58.084186Z","shell.execute_reply.started":"2024-04-10T12:30:41.162617Z","shell.execute_reply":"2024-04-10T12:30:58.082750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import deepchem as dc","metadata":{"execution":{"iopub.status.busy":"2024-04-10T11:04:45.225326Z","iopub.execute_input":"2024-04-10T11:04:45.225800Z","iopub.status.idle":"2024-04-10T11:04:45.232039Z","shell.execute_reply.started":"2024-04-10T11:04:45.225752Z","shell.execute_reply":"2024-04-10T11:04:45.230686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/leash-BELKA/train.csv', nrows=100)\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-04-10T11:04:45.234332Z","iopub.execute_input":"2024-04-10T11:04:45.234886Z","iopub.status.idle":"2024-04-10T11:04:45.287709Z","shell.execute_reply.started":"2024-04-10T11:04:45.234840Z","shell.execute_reply":"2024-04-10T11:04:45.286494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/leash-BELKA/test.csv')\ntest","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:13:11.953924Z","iopub.execute_input":"2024-04-10T12:13:11.954448Z","iopub.status.idle":"2024-04-10T12:13:18.441546Z","shell.execute_reply.started":"2024-04-10T12:13:11.954411Z","shell.execute_reply":"2024-04-10T12:13:18.439876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get molecules\nfrom rdkit import Chem\nimport pandas as pd\n\ndef smiles_to_molecule(row):\n    mol_list = []\n    for col in ['buildingblock1_smiles', 'buildingblock2_smiles', 'buildingblock3_smiles', 'molecule_smiles']: \n        smile = row[col]\n        mol = Chem.MolFromSmiles(smile)\n        if mol is None:\n            return None\n        try:\n            Chem.Kekulize(mol)\n        except:\n            pass  \n        mol_list.append(mol)\n    # Initialize a new molecule to hold the combined structure\n    combined_molecule = Chem.Mol()\n\n    # Loop over each molecule in the list and combine them into one\n    for mol in mol_list:\n        if mol:\n            combined_molecule = Chem.CombineMols(combined_molecule, mol)\n    return combined_molecule\n\ndef get_molecules(filename, rows):\n    df = pd.read_csv(filename, nrows=rows)    \n    molecule_lists = df.apply(smiles_to_molecule, axis=1)\n\n    return molecule_lists","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:08:40.828277Z","iopub.execute_input":"2024-04-10T12:08:40.829360Z","iopub.status.idle":"2024-04-10T12:08:40.839791Z","shell.execute_reply.started":"2024-04-10T12:08:40.829316Z","shell.execute_reply":"2024-04-10T12:08:40.838335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_molecules = get_molecules('/kaggle/input/leash-BELKA/train.csv', 1000)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:28:47.660876Z","iopub.execute_input":"2024-04-10T12:28:47.661789Z","iopub.status.idle":"2024-04-10T12:28:48.842725Z","shell.execute_reply.started":"2024-04-10T12:28:47.661744Z","shell.execute_reply":"2024-04-10T12:28:48.841039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import deepchem as dc\nfrom rdkit import Chem\ndef get_dataset(molecules):\n    # Featurize each molecule individually\n    featurizer = dc.feat.ConvMolFeaturizer()\n    featurized_molecules = []\n    for sublist in all_molecules:\n        featurized_sublist = featurizer.featurize(sublist)\n        featurized_molecules.append(featurized_sublist)\n\n    # Combine featurized molecules\n    combined_featurized_molecules = []\n    for sublist in featurized_molecules:\n        combined_featurized_molecules.extend(sublist)\n\n    # Convert to dataset\n    dataset = dc.data.NumpyDataset(X=combined_featurized_molecules)\n    \n    return dataset","metadata":{"execution":{"iopub.status.busy":"2024-04-10T11:40:38.887738Z","iopub.execute_input":"2024-04-10T11:40:38.888802Z","iopub.status.idle":"2024-04-10T11:40:38.897623Z","shell.execute_reply.started":"2024-04-10T11:40:38.888757Z","shell.execute_reply":"2024-04-10T11:40:38.896132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = get_dataset(all_molecules)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:35:35.313848Z","iopub.execute_input":"2024-04-10T12:35:35.314508Z","iopub.status.idle":"2024-04-10T12:35:43.000802Z","shell.execute_reply.started":"2024-04-10T12:35:35.314454Z","shell.execute_reply":"2024-04-10T12:35:42.999192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = dc.models.GraphConvModel(n_tasks=1, mode='regression', dropout=0.2)\nmodel.fit(dataset)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:35:45.510006Z","iopub.execute_input":"2024-04-10T12:35:45.510453Z","iopub.status.idle":"2024-04-10T12:36:05.325090Z","shell.execute_reply.started":"2024-04-10T12:35:45.510420Z","shell.execute_reply":"2024-04-10T12:36:05.323985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_molecules = get_molecules('/kaggle/input/leash-BELKA/test.csv', 1000)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:29:37.220134Z","iopub.execute_input":"2024-04-10T12:29:37.220551Z","iopub.status.idle":"2024-04-10T12:29:38.389704Z","shell.execute_reply.started":"2024-04-10T12:29:37.220521Z","shell.execute_reply":"2024-04-10T12:29:38.388331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = get_dataset(test_molecules)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:29:41.136210Z","iopub.execute_input":"2024-04-10T12:29:41.137378Z","iopub.status.idle":"2024-04-10T12:29:48.745279Z","shell.execute_reply.started":"2024-04-10T12:29:41.137341Z","shell.execute_reply":"2024-04-10T12:29:48.743296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:36:10.129905Z","iopub.execute_input":"2024-04-10T12:36:10.131029Z","iopub.status.idle":"2024-04-10T12:36:11.768737Z","shell.execute_reply.started":"2024-04-10T12:36:10.130988Z","shell.execute_reply":"2024-04-10T12:36:11.766888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set only first 1000 rows\ndf_sub = pd.read_csv('/kaggle/input/leash-BELKA/sample_submission.csv')\ndf_sub.loc[0:999, 'binds'] = predictions\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2024-04-10T12:36:14.099711Z","iopub.execute_input":"2024-04-10T12:36:14.100152Z","iopub.status.idle":"2024-04-10T12:36:14.605009Z","shell.execute_reply.started":"2024-04-10T12:36:14.100120Z","shell.execute_reply":"2024-04-10T12:36:14.603669Z"},"trusted":true},"execution_count":null,"outputs":[]}]}