{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8042988,"sourceType":"datasetVersion","datasetId":4740586},{"sourceId":8111844,"sourceType":"datasetVersion","datasetId":4791953}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install rdkit\n!pip install duckdb\n!pip install venny4py","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-16T20:42:14.205149Z","iopub.execute_input":"2024-06-16T20:42:14.205525Z","iopub.status.idle":"2024-06-16T20:42:56.624167Z","shell.execute_reply.started":"2024-06-16T20:42:14.205497Z","shell.execute_reply":"2024-06-16T20:42:56.622899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reference Notebooks\n* https://www.kaggle.com/code/seshurajup/eda-smiles/notebook\n#### Shrunken dataset\n* https://www.kaggle.com/code/shlomoron/belka-shrinking-the-dataset\n* https://www.kaggle.com/code/shlomoron/belka-shrunken-train-set-loading\n#### Venn comparison of BBs and unified Shrunken\n* https://www.kaggle.com/code/makio323/belka-shrunken-dataset-with-unified-smiles-index/notebook\n#### Seperation of Triazine and Non-Triazine \n* https://www.kaggle.com/code/chemdatafarmer/scaffold-exploration/notebook","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom rdkit import Chem\nfrom rdkit.Chem import Draw, AllChem\nfrom rdkit import RDLogger\nimport pickle\nimport duckdb\nfrom matplotlib_venn import venn3, venn2\nfrom venny4py.venny4py import *\nfrom tqdm import tqdm\nimport os\nimport gc","metadata":{"execution":{"iopub.status.busy":"2024-06-16T20:43:53.080303Z","iopub.execute_input":"2024-06-16T20:43:53.080683Z","iopub.status.idle":"2024-06-16T20:43:54.090487Z","shell.execute_reply.started":"2024-06-16T20:43:53.080647Z","shell.execute_reply":"2024-06-16T20:43:54.089432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/leash-BELKA/train.csv', nrows=1000)\ntrain=train[train[\"protein_name\"]==\"BRD4\"].reset_index(drop=True)\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize BBs and Product SMILES\ndef plot_mols(df, mol_count):\n    df2=df.reset_index(drop=True)\n    mols = []\n    for i in range(mol_count):\n        mols+=[Chem.MolFromSmiles(df2['buildingblock1_smiles'][i]),Chem.MolFromSmiles(df2['buildingblock2_smiles'][i]),Chem.MolFromSmiles(df2['buildingblock3_smiles'][i]),Chem.MolFromSmiles(df2['molecule_smiles'][i])]\n    return Draw.MolsToGridImage(mols, molsPerRow=4, subImgSize=(400,300))\n    \nplot_mols(train, 4)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:46:41.346664Z","iopub.execute_input":"2024-06-07T04:46:41.346991Z","iopub.status.idle":"2024-06-07T04:46:41.528745Z","shell.execute_reply.started":"2024-06-07T04:46:41.346963Z","shell.execute_reply":"2024-06-07T04:46:41.527445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* BB1: transformed part exist in Product\n* BB2-3: no change in Product\n* Some BBs: has extra parts (HBr, HCl) connected with (.)","metadata":{}},{"cell_type":"code","source":"# Get data via duckdb\n\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\ndf_bind0 = con.query(f\"\"\"SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 2000\"\"\").df()\n\ndf_bind1 = con.query(f\"\"\"SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 2000\"\"\").df()\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:46:41.530000Z","iopub.execute_input":"2024-06-07T04:46:41.530552Z","iopub.status.idle":"2024-06-07T04:47:36.571591Z","shell.execute_reply.started":"2024-06-07T04:46:41.530523Z","shell.execute_reply":"2024-06-07T04:47:36.570505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_mols(df_bind1[0:1005], 4)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:36.573080Z","iopub.execute_input":"2024-06-07T04:47:36.573460Z","iopub.status.idle":"2024-06-07T04:47:36.749521Z","shell.execute_reply.started":"2024-06-07T04:47:36.573423Z","shell.execute_reply":"2024-06-07T04:47:36.747671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_mols(df_bind1[1000:1005], 4)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:36.751258Z","iopub.execute_input":"2024-06-07T04:47:36.751666Z","iopub.status.idle":"2024-06-07T04:47:36.936993Z","shell.execute_reply.started":"2024-06-07T04:47:36.751632Z","shell.execute_reply":"2024-06-07T04:47:36.935892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con = duckdb.connect()\nbinds_train = con.query(f\"\"\"(SELECT binds, count(*) as bind_count FROM parquet_scan('{train_path}') GROUP BY binds)\"\"\").df()\n# binds_test = con.query(f\"\"\"(SELECT binds, count(*) as binds_count FROM parquet_scan('{test_path}') GROUP BY binds)\"\"\").df()\n\nprot_train = con.query(f\"\"\"(SELECT protein_name, count(*) as prot_count FROM parquet_scan('{train_path}') GROUP BY protein_name)\"\"\").df()\nprot_test = con.query(f\"\"\"(SELECT protein_name, count(*) as prot_count FROM parquet_scan('{test_path}') GROUP BY protein_name)\"\"\").df()\n\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:36.942203Z","iopub.execute_input":"2024-06-07T04:47:36.942623Z","iopub.status.idle":"2024-06-07T04:47:44.672030Z","shell.execute_reply.started":"2024-06-07T04:47:36.942570Z","shell.execute_reply":"2024-06-07T04:47:44.670710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train:\")\nprint(binds_train)\nprint(prot_train)\n\nprint(\"Test:\")\n# binds_train\nprint(prot_test)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:44.673526Z","iopub.execute_input":"2024-06-07T04:47:44.673925Z","iopub.status.idle":"2024-06-07T04:47:44.686018Z","shell.execute_reply.started":"2024-06-07T04:47:44.673893Z","shell.execute_reply":"2024-06-07T04:47:44.684397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # READ train DF from SHRUNKEN (takes few minutes)\n# dtypes = {'buildingblock1_smiles': np.int16, 'buildingblock2_smiles': np.int16, 'buildingblock3_smiles': np.int16,\n#           'binds_BRD4':np.byte, 'binds_HSA':np.byte, 'binds_sEH':np.byte}\n\n# train = pd.read_csv('/kaggle/input/belka-shrunken-train-set/train.csv', dtype = dtypes)\n# print(len(train))\n# train.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:44.687607Z","iopub.execute_input":"2024-06-07T04:47:44.688203Z","iopub.status.idle":"2024-06-07T04:47:44.704292Z","shell.execute_reply.started":"2024-06-07T04:47:44.688169Z","shell.execute_reply":"2024-06-07T04:47:44.703060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One of the shrinking strategies is encoding the BBs (building blocks) to int16 indices. To recover the original BBs, you need to use the attached dictionaries:\nBBs_dict_reverse_1 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_1.p', 'br'))\nBBs_dict_reverse_2 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_2.p', 'br'))\nBBs_dict_reverse_3 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_3.p', 'br'))\n# Test\nBBs_dict_reverse_1_test = pickle.load(open('/kaggle/input/belka-shrunken-train-set/test_dicts/BBs_dict_reverse_1_test.p', 'br'))\nBBs_dict_reverse_2_test = pickle.load(open('/kaggle/input/belka-shrunken-train-set/test_dicts/BBs_dict_reverse_2_test.p', 'br'))\nBBs_dict_reverse_3_test = pickle.load(open('/kaggle/input/belka-shrunken-train-set/test_dicts/BBs_dict_reverse_3_test.p', 'br'))\n\n\n# buildingblock3_smiles_original = [BBs_dict_reverse_3[x] for x in train.buildingblock3_smiles[:1000]]\n# print(buildingblock3_smiles_original[0])","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:44.706023Z","iopub.execute_input":"2024-06-07T04:47:44.706448Z","iopub.status.idle":"2024-06-07T04:47:44.762276Z","shell.execute_reply.started":"2024-06-07T04:47:44.706412Z","shell.execute_reply":"2024-06-07T04:47:44.761087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read BB_dictionaries\nBBs_dict_1 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_1.p', 'br'))\nBBs_dict_2 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_2.p', 'br'))\nBBs_dict_3 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_3.p', 'br'))\nBBs_dict_1_test = pickle.load(open('/kaggle/input/belka-shrunken-train-set/test_dicts/BBs_dict_1_test.p', 'br'))\nBBs_dict_2_test = pickle.load(open('/kaggle/input/belka-shrunken-train-set/test_dicts/BBs_dict_2_test.p', 'br'))\nBBs_dict_3_test = pickle.load(open('/kaggle/input/belka-shrunken-train-set/test_dicts/BBs_dict_3_test.p', 'br'))","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:44.763530Z","iopub.execute_input":"2024-06-07T04:47:44.763833Z","iopub.status.idle":"2024-06-07T04:47:44.811098Z","shell.execute_reply.started":"2024-06-07T04:47:44.763807Z","shell.execute_reply":"2024-06-07T04:47:44.809943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bb_1 = BBs_dict_1.keys()\nbb_2 = BBs_dict_2.keys()\nbb_3 = BBs_dict_3.keys()\n\nbb_1_test = BBs_dict_1_test.keys()\nbb_2_test = BBs_dict_2_test.keys()\nbb_3_test = BBs_dict_3_test.keys()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:44.812427Z","iopub.execute_input":"2024-06-07T04:47:44.812759Z","iopub.status.idle":"2024-06-07T04:47:44.819242Z","shell.execute_reply.started":"2024-06-07T04:47:44.812731Z","shell.execute_reply":"2024-06-07T04:47:44.818267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BBs_smiles = (BBs_dict_1 | BBs_dict_2 | BBs_dict_3 | BBs_dict_1_test | BBs_dict_2_test | BBs_dict_3_test).keys()\nlen(BBs_smiles)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:44.820652Z","iopub.execute_input":"2024-06-07T04:47:44.821068Z","iopub.status.idle":"2024-06-07T04:47:44.836216Z","shell.execute_reply.started":"2024-06-07T04:47:44.821029Z","shell.execute_reply":"2024-06-07T04:47:44.834902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Molecules can be represented in various SMILES notation, to make sure all BBs are unique we can canonicalize all SMILES then compare","metadata":{}},{"cell_type":"code","source":"# Count the number of unique values in all Dictionary\n\nBBs_smiles = set(BBs_dict_1.keys()) | set(BBs_dict_2.keys()) | set(BBs_dict_3.keys()) | \\\n             set(BBs_dict_1_test.keys()) | set(BBs_dict_2_test.keys()) | set(BBs_dict_3_test.keys())\n\nnum_unique_values = len(BBs_smiles)\nprint(\"Number of unique values in BBs_smiles:\", num_unique_values)\ntype(BBs_smiles)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:44.837822Z","iopub.execute_input":"2024-06-07T04:47:44.838312Z","iopub.status.idle":"2024-06-07T04:47:44.862159Z","shell.execute_reply.started":"2024-06-07T04:47:44.838269Z","shell.execute_reply":"2024-06-07T04:47:44.860671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Canonicalize all BBs then check num of unique values again\n\nunique_smiles = set()\nfor key in BBs_smiles:\n    mol = Chem.MolFromSmiles(key)\n    if mol:\n        canonical_smiles = Chem.MolToSmiles(mol)\n        unique_smiles.add(canonical_smiles)\n        # Count the number of unique canonical SMILES\n        \nnum_unique_smiles = len(unique_smiles)\nprint(\"Number of unique canonical SMILES in BBs_smiles:\", num_unique_smiles)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:44.863817Z","iopub.execute_input":"2024-06-07T04:47:44.864494Z","iopub.status.idle":"2024-06-07T04:47:45.320776Z","shell.execute_reply.started":"2024-06-07T04:47:44.864460Z","shell.execute_reply":"2024-06-07T04:47:45.319618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Verified - All bbs are unique","metadata":{}},{"cell_type":"code","source":"# plt.figure(figsize=(3, 3)) \nvenn3([bb_1, bb_2, bb_3], ('BB1', 'BB2', 'BB3'))\nplt.show()\nvenn3([bb_1_test, bb_2_test, bb_3_test], ('BB1_test', 'BB2_test', 'BB3_test'))\nplt.show()\nvenn2([bb_1, bb_1_test], ('BB1', 'BB1_test'))\nplt.show()    \n\nsets = {\n    'BB2': set(bb_2),\n    'BB3': set(bb_3),\n    'BB2_test': set(bb_2_test),\n    'BB3_test': set(bb_3_test)\n}\nvenny4py(sets=sets)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:47:45.321998Z","iopub.execute_input":"2024-06-07T04:47:45.322333Z","iopub.status.idle":"2024-06-07T04:47:46.074738Z","shell.execute_reply.started":"2024-06-07T04:47:45.322305Z","shell.execute_reply":"2024-06-07T04:47:46.073404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* BB1 is seperated from BB2 an BB3\n* BB2 and BB3 are highly overlapped\n* From the figure below it looks BB2 and BB3 interchangable\n* Can we swap BB2 and BB3 ? if it is we can represent the input accordingly.\n* To verify it, we can search DB any examples that BB1 is same but BB2xBB3 swapped examples\n* Since BB1 is not directly interacted with Protein, how much important is that?\n\n <img src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-user-content/o/inbox%2F1095143%2F1901c6caa0c6c011617f4dec525d7bbe%2FKaggle%20v2%20(1).png?generation=1712179256934503&alt=media\" alt=\"Process\" style=\"width: 800px;\"/>","metadata":{}},{"cell_type":"code","source":"# READ train DF from unified SHRUNKEN Dataset (ids should match for BB2 and BB3)\ndtypes = {'buildingblock1_smiles': np.int16, 'buildingblock2_smiles': np.int16, 'buildingblock3_smiles': np.int16,\n          'binds_BRD4':np.byte, 'binds_HSA':np.byte, 'binds_sEH':np.byte}\n\ntrain = pd.read_csv('/kaggle/input/belka-shrunken-with-unified-smiles-index/train.csv', dtype = dtypes)\nprint(len(train))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T20:46:55.673693Z","iopub.execute_input":"2024-06-16T20:46:55.674151Z","iopub.status.idle":"2024-06-16T20:52:10.588335Z","shell.execute_reply.started":"2024-06-16T20:46:55.674111Z","shell.execute_reply":"2024-06-16T20:52:10.586340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/belka-shrunken-with-unified-smiles-index/test.csv', dtype = dtypes)\nprint(len(test))\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T20:52:10.591557Z","iopub.execute_input":"2024-06-16T20:52:10.591982Z","iopub.status.idle":"2024-06-16T20:52:13.148596Z","shell.execute_reply.started":"2024-06-16T20:52:10.591944Z","shell.execute_reply":"2024-06-16T20:52:13.147450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BBs_dict = pickle.load(open('/kaggle/input/belka-shrunken-with-unified-smiles-index/BBs_dict.p', 'br'))\nBBs_dict_reverse = pickle.load(open('/kaggle/input/belka-shrunken-with-unified-smiles-index/BBs_dict_reverse.p', 'br'))\n\nlen(BBs_dict)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T20:52:13.150086Z","iopub.execute_input":"2024-06-16T20:52:13.150512Z","iopub.status.idle":"2024-06-16T20:52:13.181755Z","shell.execute_reply.started":"2024-06-16T20:52:13.150470Z","shell.execute_reply":"2024-06-16T20:52:13.180463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T20:52:13.184343Z","iopub.execute_input":"2024-06-16T20:52:13.184775Z","iopub.status.idle":"2024-06-16T20:52:13.300843Z","shell.execute_reply.started":"2024-06-16T20:52:13.184737Z","shell.execute_reply":"2024-06-16T20:52:13.299560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find row pairs that satisfy BB2=BB3 BB3=BB2  (search for random 2000 rows)\npairs = []\ntrain_1000 = train.sample(n=100, random_state=1)\nfor idx, row in tqdm(train_1000.iterrows(), total=len(train_1000)):\n    matching_rows = train[\n                       (train['buildingblock2_smiles'] == row['buildingblock3_smiles']) & \n                       (train['buildingblock3_smiles'] == row['buildingblock2_smiles'])]\n    if len(matching_rows) > 1:\n#         pairs.extend([(idx, match_idx) for match_idx in matching_rows.index if (match_idx != idx) & (row['buildingblock2_smiles']!=row['buildingblock3_smiles']) ])\n        pairs.extend([(idx, match_idx) for match_idx in matching_rows.index if (row['buildingblock2_smiles']!=row['buildingblock3_smiles']) ])\n#         pairs.extend([(idx, match_idx) for match_idx in matching_rows.index ])\nprint(\"Total:\", len(pairs))\n\n# # Display  some pairs\n# for pair in pairs[0:3]:\n#     print(\"Row pair:\", pair,\"\\n\")\n#     print(\"Row 1:\", train[[\"buildingblock1_smiles\", \"buildingblock2_smiles\",\"buildingblock3_smiles\"]].iloc[pair[0]])\n#     print(\"Row 2:\", train[[\"buildingblock1_smiles\", \"buildingblock2_smiles\",\"buildingblock3_smiles\"]].iloc[pair[1]])\n#     print()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:52:57.027869Z","iopub.execute_input":"2024-06-07T04:52:57.028198Z","iopub.status.idle":"2024-06-07T04:53:16.141442Z","shell.execute_reply.started":"2024-06-07T04:52:57.028170Z","shell.execute_reply":"2024-06-07T04:53:16.140185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* For the random 1000 rows no match is found, so we assume no BB2/BB3 swaps exist in the data\n* Still the BB2/BB3 order looks not important\n* Lets check BB2=BB3 cases in the data (should be  271*691=187261 if it is a complete set)","metadata":{}},{"cell_type":"code","source":"# Check BB2=BB3 rows 271*691 = 187261\nfiltered_df = train[(train['buildingblock2_smiles'] == train['buildingblock3_smiles'])]\nfiltered_df","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:53:16.143051Z","iopub.execute_input":"2024-06-07T04:53:16.143423Z","iopub.status.idle":"2024-06-07T04:53:16.303229Z","shell.execute_reply.started":"2024-06-07T04:53:16.143393Z","shell.execute_reply":"2024-06-07T04:53:16.301969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 187261 - 186497 = 764 rows missing (maybe in test set?) ","metadata":{}},{"cell_type":"code","source":"# Check BB2=BB3 rows 271*691 = 187261\nfiltered_test_df = test[(test['buildingblock2_smiles'] == test['buildingblock3_smiles']) & (test['buildingblock1_smiles'] < 271) & (test['buildingblock3_smiles'] < 964)]\nfiltered_test_df","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:53:16.305186Z","iopub.execute_input":"2024-06-07T04:53:16.305643Z","iopub.status.idle":"2024-06-07T04:53:16.330443Z","shell.execute_reply.started":"2024-06-07T04:53:16.305603Z","shell.execute_reply":"2024-06-07T04:53:16.329121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Yes they are in the test set, now all BB2==BB3 are complete \n* How about remaning matches in the test set?","metadata":{}},{"cell_type":"code","source":"# All matches\nfiltered_test_df = test[(test['buildingblock2_smiles'] == test['buildingblock3_smiles'])]\nfiltered_test_df","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:53:16.332060Z","iopub.execute_input":"2024-06-07T04:53:16.332416Z","iopub.status.idle":"2024-06-07T04:53:16.351968Z","shell.execute_reply.started":"2024-06-07T04:53:16.332385Z","shell.execute_reply":"2024-06-07T04:53:16.350836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train exclusive matches\nfiltered_test_df = test[(test['buildingblock2_smiles'] == test['buildingblock3_smiles']) & (test['buildingblock1_smiles'] >= 271) & (test['buildingblock3_smiles'] >= 964)]\nfiltered_test_df","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:53:16.357512Z","iopub.execute_input":"2024-06-07T04:53:16.357950Z","iopub.status.idle":"2024-06-07T04:53:16.379968Z","shell.execute_reply.started":"2024-06-07T04:53:16.357917Z","shell.execute_reply":"2024-06-07T04:53:16.378535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Interesting --->  1954-1190 = 764\n* For BB2=BB3 case all exclusive test BB1s matches with only exclusive BB2&BB3's\n* lets check for the other cases","metadata":{}},{"cell_type":"code","source":"print(\"Test size\", len(test))\n\n#only with train conditions\nfiltered_test_df_inc = test[ (test['buildingblock1_smiles'] < 271) & (test['buildingblock2_smiles'] < 1145) & (test['buildingblock3_smiles'] < 1145)]\nfiltered_test_df_inc\nprint(\"Test size (train conditions)\", len(filtered_test_df_inc))\n\n#excluding train conditions\nfiltered_test_df_exc = test[(test['buildingblock1_smiles'] >= 271) & (test['buildingblock2_smiles'] >= 1145) & (test['buildingblock3_smiles'] >= 1145)]\nfiltered_test_df_exc\nprint(\"Test size (excluding train conditions)\", len(filtered_test_df_exc))\n\nprint(len(filtered_test_df_inc), \"+\", len(filtered_test_df_exc),\"=\",(len(filtered_test_df_inc)+ len(filtered_test_df_exc)) )\n\n\ntotal_true_inc = filtered_test_df_inc[['is_BRD4', 'is_HSA', 'is_sEH']].sum().sum()\ntotal_true_exc = filtered_test_df_exc[['is_BRD4', 'is_HSA', 'is_sEH']].sum().sum()\n\nprint(\"Test size (total) (train conditions)\", total_true_inc)\nprint(\"Test size (total) (excluding train conditions)\", total_true_exc)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:53:16.381622Z","iopub.execute_input":"2024-06-07T04:53:16.381992Z","iopub.status.idle":"2024-06-07T04:53:16.468089Z","shell.execute_reply.started":"2024-06-07T04:53:16.381963Z","shell.execute_reply":"2024-06-07T04:53:16.467066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **We extracted an important information!**\n** Train_excluded test data contains exclusive for both BB1 and BB2&BB3\n** Test set contains 58% fully new BBs and 42% train BBs\n** Test set contains 34% instances from new BBs and 66% from train BBs\n** Public LB test data 35% of all but we dont know how they split it. Maybe all from 66% (train) which may cause a huge shakeup\n*** **Update:** This notebooks shows almost all public data is from %66 part https://www.kaggle.com/code/junkoda/unknowns-in-public-test-set/notebook\n** **Future Note:** It is possible to train 2 different models \n*** First one will only focus on trains shared data (we can develop non-chemistry knowledge models)\n*** Second one will be generalizable model that includes chemistry knowledge\n**** **For second model we need to have a similar splitting approach like they did for the exclusive part of the test set**\n** For now lets continue analyzing our data","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:53:16.469370Z","iopub.execute_input":"2024-06-07T04:53:16.469727Z","iopub.status.idle":"2024-06-07T04:53:16.555566Z","shell.execute_reply.started":"2024-06-07T04:53:16.469698Z","shell.execute_reply":"2024-06-07T04:53:16.554328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"triazine = Chem.MolFromSmiles('C1=NC=NC=N1')\ntriazine","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install mapply","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:53:47.117560Z","iopub.execute_input":"2024-06-07T04:53:47.118045Z","iopub.status.idle":"2024-06-07T04:54:03.060287Z","shell.execute_reply.started":"2024-06-07T04:53:47.118013Z","shell.execute_reply":"2024-06-07T04:54:03.059058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mapply\nmapply.init(n_workers=-1,progressbar=True,)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:54:03.063421Z","iopub.execute_input":"2024-06-07T04:54:03.063938Z","iopub.status.idle":"2024-06-07T04:54:03.203848Z","shell.execute_reply.started":"2024-06-07T04:54:03.063894Z","shell.execute_reply":"2024-06-07T04:54:03.202486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_for_triazine(x):\n    check = Chem.MolFromSmiles(x).HasSubstructMatch(triazine)\n    return check","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:54:03.205237Z","iopub.execute_input":"2024-06-07T04:54:03.206191Z","iopub.status.idle":"2024-06-07T04:54:03.211636Z","shell.execute_reply.started":"2024-06-07T04:54:03.206154Z","shell.execute_reply":"2024-06-07T04:54:03.210612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"triazine = Chem.MolFromSmiles('C1=NC=NC=N1')\ntriazine","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:54:47.863483Z","iopub.execute_input":"2024-06-07T04:54:47.864703Z","iopub.status.idle":"2024-06-07T04:54:47.901175Z","shell.execute_reply.started":"2024-06-07T04:54:47.864662Z","shell.execute_reply":"2024-06-07T04:54:47.899981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['triazine'] = test['molecule_smiles'].mapply(check_for_triazine)\nprint(\"Total triazines:\",len(test[test['triazine']==True]))","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:54:50.763970Z","iopub.execute_input":"2024-06-07T04:54:50.764431Z","iopub.status.idle":"2024-06-07T04:57:17.801353Z","shell.execute_reply.started":"2024-06-07T04:54:50.764401Z","shell.execute_reply":"2024-06-07T04:57:17.799810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get a dataframe where there are no triazines in the data.\nnot_triazines = test[test['triazine']==False]\nprint(\"Total Non-triazines:\",len(not_triazines))\n#Let's see what some of the non-triazine containing molecules in the test set look like.\nDraw.MolsToGridImage([Chem.MolFromSmiles(x) for x in not_triazines['molecule_smiles'].sample(12)], molsPerRow=4, subImgSize=(400,300))","metadata":{"execution":{"iopub.status.busy":"2024-06-07T04:58:48.911716Z","iopub.execute_input":"2024-06-07T04:58:48.912179Z","iopub.status.idle":"2024-06-07T04:58:49.139191Z","shell.execute_reply.started":"2024-06-07T04:58:48.912145Z","shell.execute_reply":"2024-06-07T04:58:49.137843Z"},"trusted":true},"execution_count":null,"outputs":[]}]}