{"cells":[{"metadata":{},"cell_type":"markdown","source":"this kernel is from \nhttps://www.kaggle.com/manojprabhaakr/similar-duplicate-images-in-aptos-data\nand  https://www.kaggle.com/maxwell110/duplicated-list-csv-file/ \n\nI do three things:\n1. change phash to md5 according see-'s comment https://www.kaggle.com/maxwell110/duplicated-list-csv-file/comments#575422;\n2. Duplicated with different label\n3. Duplicated in both train and test."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nprint(os.listdir(\"../input\"))\nimport sys;\nimport hashlib;\nfrom os.path import isfile\nfrom joblib import Parallel, delayed\nimport psutil","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\nprint(train_df.shape)\ntest_df = pd.read_csv(\"../input/sample_submission.csv\")\ntest_df['diagnosis'] = np.nan\ntrain = train_df.append(test_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def expand_path(p):\n    if isfile('../input/train_images/' + p + '.png'): return '../input/train_images/' + p + '.png'\n    if isfile('../input/test_images/' + p + '.png'): return '../input/test_images/' + p + '.png'\n    return p\ndef getImageMetaData(p):\n    strFile = expand_path(p)\n    file = None;\n    bRet = False;\n    strMd5 = \"\";\n    \n    try:\n        file = open(strFile, \"rb\");\n        md5 = hashlib.md5();\n        strRead = \"\";\n        \n        while True:\n            strRead = file.read(8096);\n            if not strRead:\n                break;\n            md5.update(strRead);\n        #read file finish\n        bRet = True;\n        strMd5 = md5.hexdigest();\n    except:\n        bRet = False;\n    finally:\n        if file:\n            file.close()\n\n    return p,strMd5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_meta_l = Parallel(n_jobs=psutil.cpu_count(), verbose=1)(\n    (delayed(getImageMetaData)(fp) for fp in train.id_code))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"img_meta_df = pd.DataFrame(np.array(img_meta_l))\nimg_meta_df.columns = ['id_code', 'strMd5']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.merge(img_meta_df,on='id_code')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['strMd5_count'] = train.groupby('strMd5').id_code.transform('count')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['strMd5_train_count'] = train['strMd5'].map(train.groupby('strMd5')['diagnosis'].apply(lambda x:x.notnull().sum()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['strMd5_nunique'] = train.groupby('strMd5')['diagnosis'].transform('nunique').astype('int')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.to_csv('strMd5.csv',index=None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[train.strMd5_count>1].strMd5_count.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Duplicated with same label**"},{"metadata":{"trusted":true},"cell_type":"code","source":"train[(train.strMd5_train_count>1)&(train.strMd5_nunique==1)].strMd5_count.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"strMd51 = train[(train.strMd5_count>1)&(train.strMd5_nunique==1)].strMd5.unique()\nstrMd5 = strMd51[0]\nsize = len(train[train['strMd5'] == strMd5]['id_code'])\nfig = plt.figure(figsize = (20, 5))\nfor idx, img_name in enumerate(train[train['strMd5'] == strMd5]['id_code'][:size]):\n    y = fig.add_subplot(1, size, idx+1)\n    img = cv2.imread(expand_path(img_name))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    class_id = train[train.id_code==img_name]['diagnosis'].values\n    y.set_title(img_name+f'Label: {class_id}')\n    y.imshow(img)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Duplicated with different label**"},{"metadata":{"trusted":true},"cell_type":"code","source":"train[(train.strMd5_count>1)&(train.strMd5_nunique>1)].strMd5_count.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"strMd52 = train[(train.strMd5_count>1)&(train.strMd5_nunique>1)].strMd5.unique()\nstrMd5 = strMd52[0]\nfor strMd5 in strMd52[:5]:\n    size = len(train[train['strMd5'] == strMd5]['id_code'])\n    fig = plt.figure(figsize = (20, 5))\n    for idx, img_name in enumerate(train[train['strMd5'] == strMd5]['id_code'][:size]):\n        y = fig.add_subplot(1, size, idx+1)\n        img = cv2.imread(expand_path(img_name))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        class_id = train[train.id_code==img_name]['diagnosis'].values\n        y.set_title(img_name+f'Label: {class_id}')\n        y.imshow(img)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Duplicated in both train and test**"},{"metadata":{"trusted":true},"cell_type":"code","source":"train[(train.strMd5_count>1)&(train.diagnosis.isnull())].shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"strMd52 = train[(train.strMd5_count>1)&(train.diagnosis.isnull())].strMd5.unique()\nstrMd5 = strMd52[0]\nfor strMd5 in strMd52[:5]:\n    size = len(train[train['strMd5'] == strMd5]['id_code'])\n    fig = plt.figure(figsize = (20, 5))\n    for idx, img_name in enumerate(train[train['strMd5'] == strMd5]['id_code'][:size]):\n        y = fig.add_subplot(1, size, idx+1)\n        img = cv2.imread(expand_path(img_name))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        class_id = train[train.id_code==img_name]['diagnosis'].values\n        y.set_title(img_name+f'Label: {class_id}')\n        y.imshow(img)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**About leak**"},{"metadata":{"trusted":true},"cell_type":"code","source":"train[(train.strMd5_count==2)]['strMd5_train_count'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"strMd52 = train[(train.strMd5_count>2)].strMd5.unique()\nstrMd5 = strMd52[0]\nfor strMd5 in strMd52:\n    size = len(train[train['strMd5'] == strMd5]['id_code'])\n    fig = plt.figure(figsize = (20, 5))\n    for idx, img_name in enumerate(train[train['strMd5'] == strMd5]['id_code'][:size]):\n        y = fig.add_subplot(1, size, idx+1)\n        img = cv2.imread(expand_path(img_name))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        class_id = train[train.id_code==img_name]['diagnosis'].values\n        y.set_title(img_name+f'Label: {class_id}')\n        y.imshow(img)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**conclusion**\nthere are 2×255 2-duplicated image. \n2×116 are in train. 89 have same label, 27 have differnet label;\n2×134 are in train and test,which means leak;\n2×5 are in test.\n\nthere are 3×3 3-duplicated image. \n1×3 are in train.\n2×3 are in train and test.\n\nthere are 3×4 4-duplicated image. \n3×4 are in train and test.\n\nthere are 1×5 5-duplicated image. \n1×5 are in train and test.\n\nthere are total 141(134+7) leak in test."},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}