{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8042988,"sourceType":"datasetVersion","datasetId":4740586}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook demonstrates opening the shrunken train set for the BELKA competition. We can load all the data in a Kaggle notebook with the shrunken set.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport pickle","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-06T18:01:45.037854Z","iopub.execute_input":"2024-04-06T18:01:45.038318Z","iopub.status.idle":"2024-04-06T18:01:46.225881Z","shell.execute_reply.started":"2024-04-06T18:01:45.038284Z","shell.execute_reply":"2024-04-06T18:01:46.224638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Original train set:  \nLoading only two rows. It cannot be fully loaded in a Kaggle environment because it is too large (over 50GB).","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/leash-BELKA/train.csv', nrows = 2)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:01:46.228086Z","iopub.execute_input":"2024-04-06T18:01:46.228666Z","iopub.status.idle":"2024-04-06T18:01:46.269979Z","shell.execute_reply.started":"2024-04-06T18:01:46.228632Z","shell.execute_reply":"2024-04-06T18:01:46.268710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Shrunken train set:  \nIt contains the same data as in the original set, but, shrank down to ~10GB, it can be easilty loaded in a Kaggle notebook :)","metadata":{}},{"cell_type":"code","source":"dtypes = {'buildingblock1_smiles': np.int16, 'buildingblock2_smiles': np.int16, 'buildingblock3_smiles': np.int16,\n          'binds_BRD4':np.byte, 'binds_HSA':np.byte, 'binds_sEH':np.byte}\n\ntrain = pd.read_csv('/kaggle/input/belka-shrunken-train-set/train.csv', dtype = dtypes)\nprint(len(train))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:01:46.271716Z","iopub.execute_input":"2024-04-06T18:01:46.272205Z","iopub.status.idle":"2024-04-06T18:07:30.586097Z","shell.execute_reply.started":"2024-04-06T18:01:46.272162Z","shell.execute_reply":"2024-04-06T18:07:30.584138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"One of the shrinking strategies is encoding the BBs (building blocks) to int16 indices. To recover the original BBs, you need to use the attached dictionaries:","metadata":{}},{"cell_type":"code","source":"BBs_dict_reverse_1 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_1.p', 'br'))\nBBs_dict_reverse_2 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_2.p', 'br'))\nBBs_dict_reverse_3 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_3.p', 'br'))","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:07:30.591194Z","iopub.execute_input":"2024-04-06T18:07:30.591688Z","iopub.status.idle":"2024-04-06T18:07:30.632397Z","shell.execute_reply.started":"2024-04-06T18:07:30.591647Z","shell.execute_reply":"2024-04-06T18:07:30.630875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"buildingblock3_smiles_original = [BBs_dict_reverse_3[x] for x in train.buildingblock3_smiles[:1000]]\nprint(buildingblock3_smiles_original[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:07:30.634326Z","iopub.execute_input":"2024-04-06T18:07:30.634766Z","iopub.status.idle":"2024-04-06T18:07:30.645435Z","shell.execute_reply.started":"2024-04-06T18:07:30.634730Z","shell.execute_reply":"2024-04-06T18:07:30.643858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Saving leaner dataframe and separately pickling the molecule_smiles column (this is for my personal work; you can ignore this part or use it too for faster loading in your notebooks)","metadata":{}},{"cell_type":"code","source":"train.drop(['molecule_smiles'], axis = 1).to_parquet('train_lean.parquet', index = False)\npickle.dump(train.molecule_smiles.to_numpy(), open('molecule_smiles.p', 'bw'))","metadata":{"execution":{"iopub.status.busy":"2024-04-06T18:10:17.020444Z","iopub.execute_input":"2024-04-06T18:10:17.020984Z","iopub.status.idle":"2024-04-06T18:15:13.489471Z","shell.execute_reply.started":"2024-04-06T18:10:17.020944Z","shell.execute_reply":"2024-04-06T18:15:13.487850Z"},"trusted":true},"execution_count":null,"outputs":[]}]}