{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))'''\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_DIR = '/kaggle/input/siim-isic-melanoma-classification/'\n# Location of the image dir\nimg_dir = BASE_DIR + 'jpeg/train/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#imports\nfrom matplotlib import pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"!ls -lrt /kaggle/input/siim-isic-melanoma-classification","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Loading Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\nprint(f\"Train shape: {train_df.shape}, Test shape: {test_df.shape}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Initial EDA on train and test","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check missing values on each column\ntrain_df.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def print_value_counts_columnwise(df, listOfColumns):\n    for col in listOfColumns:\n        print(df[col].value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Print count of unique values\nprint_value_counts_columnwise(train_df, ['sex','age_approx', 'anatom_site_general_challenge', 'diagnosis', 'benign_malignant','target'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Before plotting the distribution, let's fill all missing values","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.sex.fillna('na', inplace=True)\ntrain_df.age_approx.fillna('na', inplace=True)\ntrain_df.anatom_site_general_challenge.fillna('na', inplace=True)\n\n#now check if there are still missing values in any column\ntrain_df.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot frequencies for each column\ndef freq_plots(cols):\n    plt.figure(figsize = (15,10))\n    for i in range(len(cols)):\n        plt.subplot(2,3,i+1)\n        plot = sns.countplot(x = cols[i], data = train_df)\n        plt.xticks(rotation=45, horizontalalignment='right')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"freq_plots(['sex','age_approx','anatom_site_general_challenge','diagnosis','benign_malignant','target'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Relative frequencies to check how balanced the distribution is!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot relative frequencies for each column\ndef rel_freq_plots(cols):\n    plt.figure(figsize = (15,10))\n    for i in range(len(cols)):\n        value_counts = train_df[cols[i]].value_counts(normalize=True)*100\n        plt.subplot(2,3,i+1)\n        plot = sns.barplot(x = value_counts.index, y = value_counts.values, alpha=0.8)\n        plt.ylabel('Percentage %')\n        plt.xticks(rotation=45, horizontalalignment='right')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rel_freq_plots(['sex','age_approx','anatom_site_general_challenge','diagnosis','benign_malignant','target'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Let's check how target is related to each feature","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"Out of {train_df.shape[0]} samples, we have {sum(train_df['target']==1)} malignant and {sum(train_df['target']==0)} benign cases, which leads to {sum(train_df['target']==1)/train_df.shape[0] * 100} % positive class and    {sum(train_df['target']==0)/train_df.shape[0] * 100} % negative class\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check how target is distributed on sex\ntmp = train_df.groupby('sex')['target'].value_counts()\ndf = pd.DataFrame(data={'exams':tmp.values}, index=tmp.index).reset_index()\nprint(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot = sns.barplot(x='sex', y='exams', hue='target', data=df)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# target vs age_approx\ntmp = train_df.groupby('age_approx')['target'].value_counts()\ndf = pd.DataFrame(data={'exams':tmp.values}, index=tmp.index).reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot = sns.barplot(x='age_approx', y='exams', hue='target', data=df)\nplt.xticks(rotation=90)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# target vs anatom_site_general_challenge\ntmp = train_df.groupby('anatom_site_general_challenge')['target'].value_counts()\ndf = pd.DataFrame(data={'exams':tmp.values}, index=tmp.index).reset_index()\nprint(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot = sns.barplot(x='anatom_site_general_challenge', y='exams', hue='target', data=df)\nplt.xticks(rotation=90)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# target vs anatom_site_general_challenge\ntmp = train_df.groupby('diagnosis')['target'].value_counts()\ndf = pd.DataFrame(data={'exams':tmp.values}, index=tmp.index).reset_index()\nprint(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Unique patient ids in train and test\nprint(f\"Out of {len(train_df)} samples, only {len(train_df.patient_id.unique())} patient ids are unique in train set.\")\nprint(f\"Out of {len(test_df)} samples, only {len(test_df.patient_id.unique())} patient ids are unique in test set.\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Checking patient overlap in train and test","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_patient_ids_train = set(train_df.patient_id.unique())\nunique_patient_ids_test = set(test_df.patient_id.unique())\ncommon_patient_ids = unique_patient_ids_train.intersection(unique_patient_ids_test)\nprint(f\"There are totally {len(common_patient_ids)} common patient ids in train and test\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There are no overlaps between train and test! Let's check how samples are distributed across patients.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#checking distribution of images for patients\nsns.countplot(train_df.patient_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It can be seen that few patients have very large number of image samples upto 115, but almost all the patients have multiple image samples.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"{sum(train_df['target'])} tumour samples are contributed by {len(train_df.loc[train_df.target==1]['patient_id'].unique())} unique patients\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Sneak Peek into malignant samples which were taken from same patient!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp = train_df.groupby('patient_id')['target'].value_counts()\nprint(tmp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame(data={'exams':tmp.values}, index=tmp.index).reset_index()\nprint(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"multiple_samples_df = df.query('target == 1 & exams > 1')[['patient_id','exams']]\nmultiple_samples_df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's take a particular patient_id: IP_9997715 which has 3 tumour samples. Let's examine if 3 samples are diagnosed in different sections of body or they're just duplicates!!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.query('patient_id == \"IP_9997715\" & target == 1')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It's clear that those samples belong to same woman taken either at different sections at same age or at same age at different sections. Let's examine with another patient.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.query('patient_id == \"IP_9111321\" & target == 1')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Oops! It looks like this male patient's samples were taken at same age and at same section of the body. It's time to examine the image samples of this patient.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"images = list(train_df.query('patient_id == \"IP_9111321\" & target == 1')['image_name'])\nprint(images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15,10))\nfor i in range(len(images)):\n    plt.subplot(2,3,i+1)\n    img = plt.imread(os.path.join(img_dir, images[i]+'.jpg'))\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    plt.title(images[i])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Samples are all different, though taken from same patient at same age and at same section of the body.","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}