{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob\nimport re\n\nfrom PIL import Image\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom tqdm import tqdm_notebook\nimport imagehash","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_PATH = '../input/bms-molecular-translation/'\ntrain_labels = pd.read_csv(INPUT_PATH + 'train_labels.csv')\ntrain_labels['path'] = train_labels['image_id'].apply(lambda x: \n                                    INPUT_PATH + \"/train/{}/{}/{}/{}.png\".format(x[0], x[1], x[2], x))\n\ntest_labels = pd.read_csv(INPUT_PATH + 'sample_submission.csv')\ntest_labels['path'] = test_labels['image_id'].apply(lambda x: \n                                    INPUT_PATH + \"/test/{}/{}/{}/{}.png\".format(x[0], x[1], x[2], x))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def hashimg(path):\n    hash = imagehash.average_hash(Image.open(path))\n    return str(hash)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hashimg(train_labels['path'].iloc[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hashimg(train_labels['path'].iloc[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hashs = []\nfor path in tqdm_notebook(train_labels['path'].iloc[:10000]):\n    hashs.append(list(hashimg(path)))\nhashs = np.array(hashs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = test_labels['path'].iloc[120]\nids = (hashs == np.array(list(hashimg(path)))).mean(1)\nidx = np.argsort(ids)[::-1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image.open(path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image.open(train_labels['path'].iloc[idx[0]])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}