{"cells":[{"metadata":{"_uuid":"b8793296dfe80e53adeae216dfe1605b6822e5f0"},"cell_type":"markdown","source":"# From the great work of Tilii and Brian\nhttps://www.kaggle.com/c/human-protein-atlas-image-classification/discussion/72534\n\n### This kernel unite and check their findings\n### It produces an unique list of duplicates from their lists"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom PIL import Image\nfrom scipy.misc import imread\nimport tensorflow as tf\nsns.set()\nimport os\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\n\ndata_train_dir = r'../input/human-protein-atlas-image-classification/train'\nanswer_file_path = r'../input/human-protein-atlas-image-classification/train.csv'\n\nchannels = ['_yellow', '_red', '_green', '_blue']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"48ef9d42327b7e4bc50702caccd8fee5ae178750"},"cell_type":"markdown","source":"# Functions for visualization"},{"metadata":{"trusted":true,"_uuid":"f9304c7fb96eb8f88e519f052aea6bda49d9a79f"},"cell_type":"code","source":"def make_rgb_image_from_four_channels(channels: list, image_width=512, image_height=512) -> np.ndarray:\n    \"\"\"\n    It makes literally RGB image from source four channels, \n    where yellow image will be yellow color, red will be red and so on  \n    \"\"\"\n    rgb_image = np.zeros(shape=(image_height, image_width, 3), dtype=np.float)\n    yellow = np.array(Image.open(channels[0]))\n    # yellow is not added as red + bleu\n    rgb_image[:, :, 0] += yellow/2   \n    rgb_image[:, :, 2] += yellow/2\n    # loop for R,G and B channels\n    for index, channel in enumerate(channels[1:]):\n        current_image = Image.open(channel)\n        rgb_image[:, :, index] += current_image\n    \n    rgb_image = np.clip(rgb_image,0,255)\n    return rgb_image.astype(np.uint8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25211fd110c2a1c6f9ba53b2efebfa066bde09df"},"cell_type":"code","source":"df_duplicates = pd.read_csv(\"../input/duplicatesproposal/duplicates-personnal.csv\")\ndf_duplicates","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8137627f2ef69c772d2789cc8437f9db8293c3c9"},"cell_type":"markdown","source":"# Remove duplicates in duplicate list "},{"metadata":{"trusted":true,"_uuid":"bd026abc93a196dc5fbd96b3c9d8f97a1cd7bfec"},"cell_type":"code","source":"one_time_list = []\ndf_duplicate_uniq = pd.DataFrame(columns=['Keep','Remove'])\nfor i in range(len(df_duplicates)):\n    if not ( df_duplicates.iloc[i,0] in one_time_list or df_duplicates.iloc[i,1] in one_time_list):\n        one_time_list = one_time_list + [df_duplicates.iloc[i,0]] + [df_duplicates.iloc[i,1]]\n        df_duplicate_uniq = df_duplicate_uniq.append({'Keep':df_duplicates.iloc[i,0],'Remove':df_duplicates.iloc[i,1]}, ignore_index=True)\n        \n# df_duplicate_uniq.iloc[99,0] and [129,0] are malformed image\ndf_duplicate_uniq.iloc[99,0], df_duplicate_uniq.iloc[99,1] = df_duplicate_uniq.iloc[99,1], df_duplicate_uniq.iloc[99,0] \ndf_duplicate_uniq.iloc[129,0], df_duplicate_uniq.iloc[129,1] = df_duplicate_uniq.iloc[129,1], df_duplicate_uniq.iloc[129,0] \n\nprint(df_duplicate_uniq.head())\nprint(\"len(df_duplicates)\",len(df_duplicates))\nprint(\"len(df_duplicate_uniq)\",len(df_duplicate_uniq))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cf9e8514d10fcc067588ac38ff806bcc820e42b1"},"cell_type":"markdown","source":"# Displaying"},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"ad4d28dc36c079780456dd828a174d1e020986d7"},"cell_type":"code","source":"rg = np.arange(95,115,1)\nfor i in rg:\n    current_line = df_duplicate_uniq.loc[i]\n    image_names_1 = [os.path.join(data_train_dir, current_line[0]) \n                   + x + '.png' for x in channels]\n    image_names_2 = [os.path.join(data_train_dir, current_line[1]) \n                   + x + '.png' for x in channels]\n    rgb_image_1 = make_rgb_image_from_four_channels(image_names_1)\n    rgb_image_2 = make_rgb_image_from_four_channels(image_names_2)\n    fig, ax = plt.subplots(nrows = 1, ncols=2, figsize=(29,29))\n    ax[0].imshow(rgb_image_1)\n    ax[1].imshow(rgb_image_2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d98acd39c8f244db85772ed09df468c6aea9fccd"},"cell_type":"markdown","source":"# Save"},{"metadata":{"trusted":true,"_uuid":"e3a8325c76f39d4094eca14388c5178d02acf95b"},"cell_type":"code","source":"df_duplicate_uniq.to_csv(\"duplicates_final_list.csv\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}