{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import datetime\nstart_time = datetime.datetime.now()\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport torch\nimport imagehash\nimport glob\nimport matplotlib.pyplot as plt\n\nfrom tqdm.auto import  tqdm\nfrom PIL import Image","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"!unzip -j -q ../input/cassava-disease/train.zip -d train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!cp -r ../input/cassava-leaf-disease-classification/train_images/* train","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"TRAIN_DIR = './train/'\nTRAIN_CSV_PATH = '../input/cassava-leaf-disease-classification/train.csv'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(TRAIN_CSV_PATH)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"funcs = [\n        imagehash.average_hash,\n        imagehash.phash,\n        imagehash.dhash,\n        imagehash.whash,\n    ]\nimage_ids = []\nhashes = []\n\nfor path in tqdm(glob.glob(TRAIN_DIR + '*.jpg' )):\n    image = Image.open(path)\n    image_id = os.path.basename(path)\n    image_ids.append(image_id)\n    hashes.append(np.array([f(image).hash for f in funcs]).reshape(256))\n\nhashes_all = np.array(hashes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"hashes_all = torch.Tensor(hashes_all.astype(int))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sims = np.array([(hashes_all[i] == hashes_all).sum(dim=1).numpy()/256 for i in range(hashes_all.shape[0])])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"indices1 = np.where(sims > 0.9)\nindices2 = np.where(indices1[0] != indices1[1])\nimage_ids1 = [image_ids[i] for i in indices1[0][indices2]]\nimage_ids2 = [image_ids[i] for i in indices1[1][indices2]]\ndups = {tuple(sorted([image_id1,image_id2])):True for image_id1, image_id2 in zip(image_ids1, image_ids2)}\nprint('found %d duplicates' % len(dups))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"duplicate_image_ids = sorted(list(dups))\nprint(duplicate_image_ids)\n# fig, axs = plt.subplots(2, 2, figsize=(15,15))\n\n# for row in range(2):\n#         for col in range(2):\n#             img_id = duplicate_image_ids[row][col]\n#             img = Image.open(TRAIN_DIR + img_id)\n#             label =str(train.loc[train['image_id'] == img_id].label.values[0])\n#             axs[row, col].imshow(img)\n#             axs[row, col].set_title(\"image_id : \"+ img_id)\n#             axs[row, col].axis('off')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"end_time = datetime.datetime.now()\nprint(end_time - start_time)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!rm -rf train/","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}