{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8526671,"sourceType":"datasetVersion","datasetId":5088393}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install rdkit\n!pip install duckdb","metadata":{"execution":{"iopub.status.busy":"2024-05-27T01:53:09.105465Z","iopub.execute_input":"2024-05-27T01:53:09.105832Z","iopub.status.idle":"2024-05-27T01:54:03.461287Z","shell.execute_reply.started":"2024-05-27T01:53:09.105797Z","shell.execute_reply":"2024-05-27T01:54:03.459940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2024-06-03T10:33:30.889730Z","iopub.execute_input":"2024-06-03T10:33:30.890130Z","iopub.status.idle":"2024-06-03T10:33:30.935353Z","shell.execute_reply.started":"2024-06-03T10:33:30.890068Z","shell.execute_reply":"2024-06-03T10:33:30.934157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\nfrom rdkit import Chem\nfrom rdkit.Chem import MACCSkeys\nfrom rdkit.Chem import AllChem\nfrom rdkit.Chem import rdmolops\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport torch","metadata":{"execution":{"iopub.status.busy":"2024-05-27T01:55:13.386044Z","iopub.execute_input":"2024-05-27T01:55:13.387271Z","iopub.status.idle":"2024-05-27T01:55:16.765897Z","shell.execute_reply.started":"2024-05-27T01:55:13.387218Z","shell.execute_reply":"2024-05-27T01:55:16.764974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample data from train dataset. \n\nTo maintain balance in the data structure, we randomly sample the same number of examples from each binds and nonbinds.","metadata":{}},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\n\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf_sample = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 1589906\n                       ) UNION ALL\n                       (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                       )\n                        \"\"\").df()\n\ncon.close()\n\ndf_sample.to_csv(\"leash_train526.csv\") \n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T01:55:40.328594Z","iopub.execute_input":"2024-05-27T01:55:40.329081Z","iopub.status.idle":"2024-05-27T01:57:56.630445Z","shell.execute_reply.started":"2024-05-27T01:55:40.329042Z","shell.execute_reply":"2024-05-27T01:57:56.628980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_ecfp(molecule, radius=4, bits=1024):\n    if molecule is None:\n        return None\n    return list(AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_chunk = True  \ndone_chunk =0\n\nf_train = open('/kaggle/working/leash_train.json', 'w')\n\nfor df in pd.read_csv(\"/kaggle/working/leash_train526.csv\", chunksize=100000):\n    df['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n    df['ecfp'] = df['molecule'].apply(generate_ecfp)\n\n    one_hot = pd.get_dummies(df['protein_name'],prefix= 'protein_name')\n    df = pd.concat([df, one_hot], axis=1)\n\n    for column in one_hot.columns:\n        df[column] = df[column].astype(int)\n    \n    df['protein_onehot'] = df[one_hot.columns].values.tolist()\n    \n    df.drop(columns=['id', 'buildingblock1_smiles', 'buildingblock2_smiles',\n       'buildingblock3_smiles', 'molecule_smiles', 'protein_name',\n       'molecule','protein_name_BRD4', 'protein_name_HSA', 'protein_name_sEH'], inplace=True)\n        \n    json_str = df.to_json(orient='records', lines=True)\n    \n    done_chunk += 100000\n    print(\"chunk_time\", done_chunk)\n    del df\n    \n    if first_chunk:\n        f_train.write(json_str) \n        first_chunk = False\n    else:\n        f_train.write('\\n' + json_str) ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport torch\n\n# Path to your JSON file\njson_file_path = '/kaggle/input/leash-train/leash_train_526.json'\n\n# Define chunk size\nchunk_size = 100000\n\n# Initialize empty lists for X and y\nX_chunks = []\ny_chunks = []\ndone_chunk = 0\n# Read the JSON file in chunks\nfor chunk in pd.read_json(json_file_path, orient='records', lines=True, chunksize=chunk_size):\n    # Process the chunk to extract features and labels\n    y_chunk = torch.tensor(chunk['binds'].tolist(), dtype=torch.float)\n    X_chunk = torch.tensor([list(ecfp) + protein for ecfp, protein in zip(chunk['ecfp'].tolist(), chunk['protein_onehot'].tolist())], dtype=torch.float32)\n    \n    # Append the chunk data to the lists\n    X_chunks.append(X_chunk)\n    y_chunks.append(y_chunk)\n    done_chunk += chunk_size\n    print(done_chunk)\n    \n    del chunk\n    del X_chunk\n    del y_chunk \n\n# Concatenate the chunks along the first dimension to create the final tensors\nX = torch.cat(X_chunks, dim=0)\ny = torch.cat(y_chunks, dim=0)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_chunks\ndel y_chunks","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split the dataset into train and test ","metadata":{}},{"cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test dataset preprocessing","metadata":{}},{"cell_type":"code","source":"test_path = '/kaggle/input/leash-BELKA/test.csv'\n\nfirst_chunk = True  \ndone_chunk =0\n\nf_test = open('/kaggle/working/leash_test527.json', 'w')\n\nfor df in pd.read_csv(test_path, chunksize=100000):\n    df['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n    df['ecfp'] = df['molecule'].apply(generate_ecfp)\n\n    one_hot = pd.get_dummies(df['protein_name'],prefix= 'protein_name')\n    df = pd.concat([df, one_hot], axis=1)\n\n    for column in one_hot.columns:\n        df[column] = df[column].astype(int)\n    \n    df['protein_onehot'] = df[one_hot.columns].values.tolist()\n    \n    df.drop(columns=['id', 'buildingblock1_smiles', 'buildingblock2_smiles',\n       'buildingblock3_smiles', 'molecule_smiles', 'protein_name',\n       'molecule','protein_name_BRD4', 'protein_name_HSA', 'protein_name_sEH'], inplace=True)\n        \n    json_str = df.to_json(orient='records', lines=True)\n    \n    done_chunk += 100000\n    print(\"chunk_time\", done_chunk)\n    del df\n        # Write the JSON string to the file\n    if first_chunk:\n        f_test.write(json_str)  # Write as is for the first chunk\n        first_chunk = False\n    else:\n        f_test.write('\\n' + json_str) ","metadata":{"execution":{"iopub.status.busy":"2024-05-27T02:20:11.180368Z","iopub.execute_input":"2024-05-27T02:20:11.181925Z","iopub.status.idle":"2024-05-27T03:04:40.614301Z","shell.execute_reply.started":"2024-05-27T02:20:11.181860Z","shell.execute_reply":"2024-05-27T03:04:40.613122Z"},"trusted":true},"execution_count":null,"outputs":[]}]}