{"cells":[{"metadata":{},"cell_type":"markdown","source":"Part 2 of [notebook](https://www.kaggle.com/keremt/cassava-eda-imagehash-cnn-dedup-old-and-new-data)\n\nIn this notebook we find dedups using Rapids"},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -qqU fastai==2.1.7","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import fastai; print(\"fastai:\", fastai.__version__)\nimport torch; print(\"torch:\", torch.__version__)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from fastai.vision.all import *\nimport torchvision ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"new_data_path = Path(\"../input/cassava-leaf-disease-classification/\")\nold_data_path = Path(\"../input/cassavaold/\")\ndedup_path = Path(\"../input/cassavadedup/\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Data"},{"metadata":{},"cell_type":"markdown","source":"### a) New Data"},{"metadata":{"trusted":true},"cell_type":"code","source":"new_data_path.ls().map(lambda o: o.name)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images = get_image_files(new_data_path/'train_images')\ntest_images = get_image_files(new_data_path/'test_images')\ntrain_df = pd.read_csv(new_data_path/'train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_images), len(test_images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_images = train_images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labeldict = json.loads((new_data_path/'label_num_to_disease_map.json').open().read())\nlabeldict = {int(k):v for k,v in labeldict.items()}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['label'].map(labeldict).value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%timeit\nimg1 = PILImage.create(train_images[0]) \nimg1 = ToTensor()(img1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%timeit\nimg2 = torchvision.io.read_image(train_images[0].as_posix())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### b) Old Data\n\nPlease upvote if you use: https://www.kaggle.com/keremt/cassavaold"},{"metadata":{"trusted":true},"cell_type":"code","source":"old_train_images = get_image_files(old_data_path/'train')\nold_test_images = get_image_files(old_data_path/'test')\nold_unsup_images = get_image_files(old_data_path/'extraimages')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"old_images = old_train_images + old_test_images + old_unsup_images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(old_images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"## CNN Based Dedup (Part2)"},{"metadata":{},"cell_type":"markdown","source":"### Normalize labels "},{"metadata":{"trusted":true},"cell_type":"code","source":"len(old_images), len(new_images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oldlabeldict = {'cbsd': 'Cassava Brown Streak Disease (CBSD)',\n                 'healthy': 'Healthy',\n                 'cmd': 'Cassava Mosaic Disease (CMD)',\n                 'cgm': 'Cassava Green Mottle (CGM)',\n                 'cbb': 'Cassava Bacterial Blight (CBB)',\n                 '0': 'Unsup', # test\n                 'extraimages': 'Unsup'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labeldict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"old_images2labels = dict(zip(old_images, [oldlabeldict[o] for o in old_images.map(lambda o: o.parent.name)]))\n\nnew_images2labels = dict(zip(train_df['image_id'], train_df['label']))\nnew_images2labels = {k:labeldict[v] for k,v in new_images2labels.items()}\nnew_images2labels = {o:new_images2labels[o.name] for o in new_images}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Counter(old_images2labels.values())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Counter(new_images2labels.values())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_images2label = {**old_images2labels, **new_images2labels}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Counter(all_images2label.values())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(all_images2label)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_vocab = {'Cassava Bacterial Blight (CBB)':0,\n             'Cassava Brown Streak Disease (CBSD)':1,\n             'Cassava Green Mottle (CGM)':2,\n             'Cassava Mosaic Disease (CMD)':3,\n             'Healthy':4, \n             'Unsup':5}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# create dataframe for merged data\nfnames, labels = zip(*all_images2label.items())\nfnames = [str(o) for o in fnames]\ndata_df = pd.DataFrame({'fnames':fnames, 'labels':labels})\ndata_df['source'] = data_df.fnames.apply(lambda o: o.split(\"/\")[2])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_df.to_csv(\"merged_training_data.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"counts_df = data_df.groupby(['source', 'labels']).count(); counts_df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We can see label distributions are different in 2 datasets"},{"metadata":{"trusted":true},"cell_type":"code","source":"counts_df.groupby(level=0).apply(lambda x: x / float(x.sum()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"counts_df.drop(index='Unsup', level='labels').apply(lambda x: x / float(x.sum()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 2) Get Embeddings "},{"metadata":{"trusted":true},"cell_type":"code","source":"all_images = old_images + new_images; len(all_images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Fastai\nsize = (224,224)\nbs = 64\n\ntfms = [[PILImage.create, ToTensor, \n         Resize(size, method='squish')], # We don't want to crop different parts of the same image if we are going to look for dedups!\n        [lambda o: all_images2label[o], Categorize(label_vocab)]]\n\ndsets = Datasets(all_images, tfms=tfms, splits=None)\n\nbatch_tfms = [IntToFloatTensor, Normalize.from_stats(*imagenet_stats)]\ndls = dsets.dataloaders(bs=bs, after_batch=batch_tfms)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_image(dsets[0][0]);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ndls.show_batch(max_n=25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load\nembeddings = torch.load(dedup_path/\"embeddings.pth\")\nembeddings.shape, len(all_images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_images = pd.read_pickle(dedup_path/\"all_images_filenames.pkl\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embeddings.shape, len(all_images)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 3) KNN"},{"metadata":{"trusted":true},"cell_type":"code","source":"import cuml","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embeddings_np = embeddings.numpy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"KNN = 4\nmodel = cuml.neighbors.NearestNeighbors(n_neighbors=KNN)\nmodel.fit(embeddings_np)\ndistances, indices = model.kneighbors(embeddings_np)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.hist(np.min(distances[:, 1:], 1));","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We have enough variability between 30 - 35 distances indicating different samples, so we can probably thresholds as < 30. You can also play with different `upper` and `lower` to see how nearest neighbors change.\n\nHere I keep 30 to be conservative, but you may pick lower upper threshold.\n\nActually let's go with 25, as we can start seeing duplicates from that point :)"},{"metadata":{"trusted":true},"cell_type":"code","source":"lower = 0\nupper = 25\nmask = (distances < upper)*(distances >= lower)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dup_idxs = mask[:, 1:].sum(1) > 0\nprint(f\"Total potential duplicates: {sum(dup_idxs)}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dup_indices, dup_mask, dup_distances = indices[dup_idxs], mask[dup_idxs], distances[dup_idxs]\nsortidxs = np.argsort(dup_distances[:, 1])\ndup_indices, dup_mask, dup_distances = dup_indices[sortidxs], dup_mask[sortidxs], dup_distances[sortidxs]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 0\nfor idxs, m in zip(dup_indices, dup_mask):\n    masked_idxs = idxs[m]\n    \n    if masked_idxs[0] != idxs[0]:\n        masked_idxs = [idxs[0]] + list(masked_idxs)\n    \n    fnames = [all_images[i] for i in masked_idxs]\n    titles = [all_images2label[fn] for fn in fnames]\n    imgs = [PILImage.create(fn) for fn in fnames]\n    show_images(imgs, imsize=5, titles=titles)\n    i += 1\n    if i == 20: break","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Save groups"},{"metadata":{"trusted":true},"cell_type":"code","source":"data_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image2groupid = {}\ni = 0\nfor idxs, m in zip(dup_indices, dup_mask):\n    \n    masked_idxs = idxs[m]\n    \n    if masked_idxs[0] != idxs[0]:\n        masked_idxs = [idxs[0]] + list(masked_idxs)\n        \n    for idx in masked_idxs:\n        image2groupid[str(all_images[idx])] = i \n    \n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(image2groupid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_df['knn_groups'] = data_df['fnames'].map(image2groupid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_groups = np.unique(data_df.dropna().knn_groups.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"k = np.random.choice(unique_groups)\ngroup_df = data_df.query(f\"knn_groups == {k}\")\ngroup_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_images([open_image(o) for o in group_df['fnames']], imsize=7)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_df.to_csv(\"merged_training_data.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}