{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob\nimport re\n\nfrom PIL import Image\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_PATH = '../input/bms-molecular-translation/'\ntrain_labels = pd.read_csv(INPUT_PATH + 'train_labels.csv')\ntrain_labels['path'] = train_labels['image_id'].apply(lambda x: \n                                    INPUT_PATH + \"/train/{}/{}/{}/{}.png\".format(x[0], x[1], x[2], x))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = train_labels.sort_values(by='InChI')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_labels['InChI'].iloc[0])\nImage.open(train_labels['path'].iloc[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_labels['InChI'].iloc[1])\nImage.open(train_labels['path'].iloc[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['InChI_len'] = train_labels['InChI'].apply(len)\ntrain_labels['InChI_SplitCount'] = train_labels['InChI'].apply(lambda x: len(x.split('/')))\ntrain_labels = train_labels[train_labels['InChI_SplitCount'] > 3]\n\ntrain_labels['InChI_1'] = train_labels['InChI'].apply(lambda x: x.split('/')[0])\ntrain_labels['InChI_2'] = train_labels['InChI'].apply(lambda x: x.split('/')[1])\ntrain_labels['InChI_3'] = train_labels['InChI'].apply(lambda x: x.split('/')[2])\ntrain_labels['InChI_4'] = train_labels['InChI'].apply(lambda x: x.split('/')[3])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"re.findall(r\"[^\\W\\d_]+|\\d+\", \"C10H12F2N2O3\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['InChI_2'].iloc[:10].apply(lambda x: re.findall(r\"[^\\W\\d_]+|\\d+\", x))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}