{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import glob\nimport cv2\nfrom PIL import Image\nimport matplotlib.pyplot as plt \nfrom tqdm import tqdm, tqdm_notebook\nimport gc\nimport seaborn as sns\nsns.set_style(\"dark\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#Image data\nimport os\nimage_train_dir = '../input/siim-isic-melanoma-classification/jpeg/train'\nlisttrain = os.listdir(image_train_dir) # dir is your directory path\nnumber_files_train = len(listtrain)\nprint('Number of images in Train dataset:',number_files_train)\n\nimage_test_dir = '../input/siim-isic-melanoma-classification/jpeg/test'\nlisttest = os.listdir(image_test_dir) # dir is your directory path\nnumber_files_test = len(listtest)\nprint('Number of images in Test dataset:',number_files_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# **Looking at some images in the Training dataset**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"img_names = glob.glob('../input/siim-isic-melanoma-classification/jpeg/train/*.jpg')\n\nfig, ax = plt.subplots(4, 4, figsize=(20, 20))\n\nfor i in range(16):\n    x = i // 4\n    y = i % 4\n    \n    path = img_names[i]\n    image_id = path.split(\"/\")[5][:-4]\n    \n    target = train_df.loc[train_df['image_name'] == image_id, 'target'].tolist()[0]\n    \n    img = Image.open(path)\n    \n    ax[x, y].imshow(img)\n    ax[x, y].axis('off')\n    ax[x, y].set_title(f'ID: {image_id}, Target: {target}')\n\nfig.suptitle(\"Training set samples\", fontsize=15)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Now, we will look at Malignent and Benign cases separately within the training images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"path_train = '../input/siim-isic-melanoma-classification/jpeg/train/'\n\n\ntrain_df_malignent = train_df[train_df['benign_malignant'] == 'malignant']\ntrain_df_benign = train_df[train_df['benign_malignant'] == 'benign']\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Displaying Benign images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(4, 4, figsize=(20, 20))\n\nfor i in range(16):\n    x = i // 4\n    y = i % 4\n    image_id = train_df_benign.iloc[i,0]\n    path = path_train + image_id + '.jpg'\n    image_id = path.split(\"/\")[5][:-4]\n    \n    \n    img = Image.open(path)\n    \n    ax[x, y].imshow(img)\n    ax[x, y].axis('off')\n    ax[x, y].set_title(f'ID: {image_id}')\n\nfig.suptitle(\"Benign samples\", fontsize=15)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Displaying Malignent images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(4, 4, figsize=(20, 20))\n\nfor i in range(16):\n    x = i // 4\n    y = i % 4\n    image_id = train_df_malignent.iloc[i,0]\n    path = path_train + image_id + '.jpg'\n    image_id = path.split(\"/\")[5][:-4]\n    \n    \n    img = Image.open(path)\n    \n    ax[x, y].imshow(img)\n    ax[x, y].axis('off')\n    ax[x, y].set_title(f'ID: {image_id}')\n\nfig.suptitle(\"Malignent samples\", fontsize=15)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Just by looking at some of these above images, it looks in like Benign cases the contrast between the background and the dark patches is less compared to Malignent cases. Also, the patch area is relatively larger in cases of Malignent compared to benign (we need to keep in mind that the images may have different zoom level, hence we cannot draw any conclusion with patch size)","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Let's dive deeper into the images","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"Loading the images in a stack to do a Pixel level exploration","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"\nimage_names = glob.glob('../input/siim-isic-melanoma-classification/jpeg/train/*.jpg')\nsize_array_train = []\n\nfor image_name in tqdm(image_names):\n    path = image_name\n    img = Image.open(path)\n    temp = img.size\n    size_array_train.append(temp)\n    \nlen(size_array_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_names = glob.glob('../input/siim-isic-melanoma-classification/jpeg/test/*.jpg')\nsize_array_test = []\n\nfor image_name in tqdm(image_names):\n    path = image_name\n    img = Image.open(path)\n    temp = img.size\n    size_array_test.append(temp)\n    \nlen(size_array_test)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's look at the Training images first","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"size_df_train = pd.DataFrame(np.row_stack(size_array_train))\nsize_df_train.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"What about test images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"size_df_test = pd.DataFrame(np.row_stack(size_array_test))\nsize_df_test.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Looks like the mean has shifted a bit from training to Testing\n* The mean and max are the same\n* The deviations in image sizes are large but the deviation is consistent for training and testing data","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Let's look at the distribution of the image sizes**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(10, 5))\n\ng1 = sns.distplot(size_df_train[0], ax=ax[0])\nax[0].set_title(\"training set\")\n\ng2 = sns.distplot(size_df_test[0], ax=ax[1])\nax[1].set_title(\"test set\")\n\n# g1.set_xticklabels(g1.get_xticklabels(), rotation=45)\n# g2.set_xticklabels(g2.get_xticklabels(), rotation=45)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(10, 5))\n\ng1 = sns.distplot(size_df_train[1], ax=ax[0])\nax[0].set_title(\"training set\")\n\ng2 = sns.distplot(size_df_test[1], ax=ax[1])\nax[1].set_title(\"test set\")\n\n# g1.set_xticklabels(g1.get_xticklabels(), rotation=45)\n# g2.set_xticklabels(g2.get_xticklabels(), rotation=45)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Distribution of height and width of the Training and Testing data looks same**","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"*Lets see if the image sizes are similar across the different classes of data*","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_malignent.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_names = glob.glob('../input/siim-isic-melanoma-classification/jpeg/train/*.jpg')\nsize_array_train = []\ni = 0\n\nfor image_name in tqdm(image_names):\n    temp = image_name.split('/')\n    temp1 = temp[-1]\n    temp2 = temp1.split('.')\n#     print(temp2[0])\n    if(train_df_benign['image_name'].str.contains(temp2[0]).any()):\n        print(temp2)\n    print(temp2[0])\n    i = i + 1\n    if(i == 4):\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_names = glob.glob('../input/siim-isic-melanoma-classification/jpeg/train/*.jpg')\nsize_array_train = []\ni = 0\n\nfor image_name in tqdm(image_names):\n    temp = image_name.split('/')\n    temp1 = temp[-1]\n    temp2 = temp1.split('.')\n#     print(temp2[0])\n    if(train_df_benign['image_name'].str.contains(temp2[0]).any()):\n        print(temp2)\n    print(temp2[0])\n    i = i + 1\n    if(i == 4):\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nimage_names = glob.glob('../input/siim-isic-melanoma-classification/jpeg/train/*.jpg')\nsize_array_train_benign = []\n\nfor image_name in tqdm(image_names):\n    path = image_name\n    temp = image_name.split('/')\n    temp1 = temp[-1]\n    temp2 = temp1.split('.')\n    if(train_df_benign['image_name'].str.contains(temp2[0]).any()):\n        img = Image.open(path)\n        temp = img.size\n        size_array_train_benign.append(temp)\n    \nlen(size_array_train_benign)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nimage_names = glob.glob('../input/siim-isic-melanoma-classification/jpeg/train/*.jpg')\nsize_array_train_malignent = []\n\nfor image_name in tqdm(image_names):\n    path = image_name\n    temp = image_name.split('/')\n    temp1 = temp[-1]\n    temp2 = temp1.split('.')\n    if(train_df_malignent['image_name'].str.contains(temp2[0]).any()):\n        img = Image.open(path)\n        temp = img.size\n        size_array_train_malignent.append(temp)\n    \nlen(size_array_train_malignent)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}