{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Preprocessing","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"This is the preprocessing notebook related to the following notebook:\n\n\n[CNN+RNN]CNN pretraining w/ regression -TRAIN","metadata":{}},{"cell_type":"markdown","source":"Many thanks to [Y.Nakama](https://www.kaggle.com/yasufuminakama) who shared great starter notebooks. This notebook uses the preprocessing notebook by Y.Nakama to extract the number of atoms of each element","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport re\nimport numpy as np\nimport pandas as pd\nfrom collections import OrderedDict\nfrom tqdm.autonotebook import tqdm","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_pickle('../input/inchi-preprocess-2/train2.pkl')\n\ndef get_train_file_path(image_id):\n    return \"../input/bms-molecular-translation/train/{}/{}/{}/{}.png\".format(\n        image_id[0], image_id[1], image_id[2], image_id \n    )\n\ntrain['file_path'] = train['image_id'].apply(get_train_file_path)\n\nprint(f'train.shape: {train.shape}')\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nplt.imshow(cv2.imread(train.loc[2, 'file_path'])[..., ::-1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['formula'] = train['InChI_text'].str.split(\"/\", expand=True, n=1).drop(1, axis=1).rename(columns={0: \"formula\"})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[['formula', 'file_path', 'InChI_length']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"atoms = ['B', 'Br', 'C', 'Cl', 'F', 'H', 'I', 'N', 'O', 'P', 'S', 'Si']\nfor atom in atoms:\n    train[atom] = -1\n\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_counts(row):    \n    counts = OrderedDict((atom, 0) for atom in atoms)\n    formula = row.formula.strip().split()\n    for i, item in enumerate(formula):\n        if item not in atoms: continue\n        elif (i + 1) == len(formula):\n            counts[item] = 1\n        elif formula[i + 1] in atoms:\n            counts[item] = 1\n        else:\n            counts[item] = int(formula[i + 1])\n    return list(counts.values())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()\nout = train.progress_apply(get_counts, axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.iloc[:, 3:] = np.stack(out.values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.sample(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_pickle(\"cnn_pretrain.pkl\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_pickle(\"cnn_pretrain.pkl\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}