{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Import required Libraries","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom matplotlib import pyplot as plt\nimport seaborn as sb","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Read Data","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"input_dir = '/kaggle/input'\nisic_dir = os.path.join(input_dir, 'siim-isic-melanoma-classification')\njpg_dir = os.path.join(isic_dir, 'jpeg')\nwork_dir = os.path.abspath(os.getcwd())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = pd.read_csv(os.path.join(isic_dir, 'train.csv'))\ntest_csv = pd.read_csv(os.path.join(isic_dir, 'test.csv'))\n\ntrain_csv.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_csv.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def preprocess(data):\n    data['sex'] = data['sex'].astype('category')\n    data['site'] = data['anatom_site_general_challenge'].astype('category')\n    data['diagnosis'] = data['diagnosis'].astype('category')\n    data['benign_malignant'] = data['benign_malignant'].astype('category')\n    \n    return data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = preprocess(train_csv)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Explore Data","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Check Missing Values","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"total = train_csv.shape[0]\nprint('Missing sex info: ', total - train_csv[train_csv.sex.notnull()].shape[0])\nprint('Missing age info: ', total - train_csv[train_csv.age_approx.notnull()].shape[0])\nprint('Missing site info: ', total - train_csv[train_csv.site.notnull()].shape[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Check if missing has any relationship with malign/benign","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv['has_missing'] = train_csv.apply(lambda row: row.isna().any(), axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sb.distplot(train_csv[train_csv.has_missing==True]['target'], label='Missing', kde=False)\nsb.distplot(train_csv[train_csv.has_missing==False]['target'], label='Full', kde=False)\nplt.legend(loc='best')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"missing_malignant = train_csv[(train_csv.has_missing==True) & (train_csv.target==1)].shape[0]\nmissing_benign = train_csv[(train_csv.has_missing==False) & (train_csv.target==1)].shape[0]\n\nprint('Malignant with missing values: ', missing_malignant)\nprint('Benign with missing values: ', missing_benign)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Check for duplicates","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_patients = train_csv['patient_id'].unique().shape[0] \nunique_images = train_csv['image_name'].unique().shape[0]\ntotal = train_csv['patient_id'].shape[0]\n\nprint('Unique patients: ', unique_patients)\nprint('Unique images: ', unique_images)\nprint('Total: ', total)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# for train and test\ncommon_images = test_csv['image_name'].isin(train_csv['image_name']).value_counts()\ncommon_patients = test_csv['patient_id'].isin(train_csv['patient_id']).value_counts()\nprint('Common images ---> ', common_images)\nprint('Common Patients ---> ', common_patients)\nprint('Total test: ', test_csv.shape[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Looks like 2056 patients have 33126 images. Let's see the distribution of images per person. Count of patient ID should do the job.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sb.distplot(train_csv[\"patient_id\"].value_counts())\nplt.title(\"Patient ID distribution\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Histograms for each variables","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# For numeric columns\nplt.rcParams['figure.figsize'] = 10,4\ntrain_csv['age_approx'].hist()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# For non-numeric columns\nplt.rcParams['figure.figsize'] = 16,12\nfig, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2,2)\ntrain_csv['sex'].value_counts().plot(kind='bar', ax=ax1, title='Sex')\ntrain_csv['anatom_site_general_challenge'].value_counts().plot(kind='bar', ax=ax2, title='Site')\ntrain_csv['diagnosis'].value_counts().plot(kind='bar', ax=ax3, title='Diagnosis')\ntrain_csv['benign_malignant'].value_counts().plot(kind='bar', ax=ax4, title='Benign/Malignant')\nplt.tight_layout() \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Finding from histogram\nWe can see that the dataset is highly imbalanced with very few malignant cases. We have to be careful while selecting optimization metric. Using accuracy might give us misleading result. ROC curve and PR curve will do a better job for selecting optimum model.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### Plot Across features","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# plt.rcParams['figure.figsize'] = 6,4\n\nplt.rcParams['figure.figsize'] = 16,6\nfig, (ax1, ax2) = plt.subplots(1,2)\nsb.countplot(x='sex', hue='benign_malignant', data=train_csv, ax=ax1)\nsb.countplot(x='anatom_site_general_challenge', hue='benign_malignant', data=train_csv, ax=ax2)\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sex_site_result = train_csv.groupby(['sex', 'site', 'benign_malignant']).agg(occurence=('target','count'))\nsex_site_result = sex_site_result.reset_index()\nsex_site_result","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sex_site_result.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sb.scatterplot(x='site', y='occurence', hue='benign_malignant', style='sex', data=sex_site_result)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sb.scatterplot(x='age_approx', y='site', hue='benign_malignant', style='sex', alpha=0.6, markers=['P', 'X'], data=train_csv)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Looks like females above 80 and males above 60 are more vulnerable and it is less likely to appear in the oral/genital area. Let's have a look at age.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# sb.pairplot(train_csv[['age_approx','benign_malignant']], hue=\"benign_malignant\")\nsb.distplot(train_csv[train_csv.target==0]['age_approx'], label='Benign')\nsb.distplot(train_csv[train_csv.target==1]['age_approx'], label='Malignant')\nplt.legend(loc='best')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sb.distplot(train_csv[train_csv.sex=='male']['age_approx'], label='Male')\nsb.distplot(train_csv[train_csv.sex=='female']['age_approx'], label='Female')\nplt.legend(loc='best')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sb.distplot(train_csv[(train_csv.sex=='male') & (train_csv.target==0)]['age_approx'], label='Male Benign')\nsb.distplot(train_csv[(train_csv.sex=='male') & (train_csv.target==1)]['age_approx'], label='Male Malignant')\nsb.distplot(train_csv[(train_csv.sex=='female') & (train_csv.target==0)]['age_approx'], label='Female Benign')\nsb.distplot(train_csv[(train_csv.sex=='female') & (train_csv.target==1)]['age_approx'], label='Female Malignant')\nplt.legend(loc='best')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"From these plots we can conclude that:\n1. Male and Female Benign distribution is similar\n2. Malignant type is mostly prevalent in higher age group\n3. Males over 60 are particularly more vulnerable.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Compare images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"benign = train_csv[train_csv.benign_malignant=='benign']\nmalignant = train_csv[train_csv.benign_malignant=='malignant']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bimg = plt.imread(jpg_dir + '/train/' + benign.iloc[0]['image_name'] + '.jpg')\nplt.imshow(bimg)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mimg = plt.imread(jpg_dir + '/train/' + malignant.iloc[0]['image_name'] + '.jpg')\nplt.imshow(mimg)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_rgb_histogram(image, ax=None):\n    if ax is None:\n        plt.hist(image[:, :, 0].ravel(), bins = 256, color = 'Red', alpha = 0.5, label='Red')\n        plt.hist(image[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5, label='Green')\n        plt.hist(image[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5, label='Blue')\n        plt.xlabel('Intensity')\n        plt.ylabel('Count')\n        plt.legend(loc='best')\n        plt.show()\n    else:\n        ax.hist(image[:, :, 0].ravel(), bins = 256, color = 'Red', alpha = 0.5, label='Red')\n        ax.hist(image[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5, label='Green')\n        ax.hist(image[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5, label='Blue')\n        ax.set_xlabel('Intensity')\n        ax.set_ylabel('Count')\n        ax.legend(loc='best')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_rgb_histogram(bimg)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_rgb_histogram(mimg)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Randomly pick some images and show their histogram","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"random_benign = benign.sample(6)\nrandom_malignant = malignant.sample(6)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"plt.rcParams['figure.figsize'] = 10,6\nfor index, row in random_benign.iterrows():\n    img = plt.imread(jpg_dir + '/train/' + row['image_name'] + '.jpg')\n    fig, (ax1, ax2) = plt.subplots(1, 2)\n    ax1.imshow(img)\n    show_rgb_histogram(img, ax2)\n    plt.tight_layout()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"for index, row in random_malignant.iterrows():\n    img = plt.imread(jpg_dir + '/train/' + row['image_name'] + '.jpg')\n    fig, (ax1, ax2) = plt.subplots(1, 2)\n    ax1.imshow(img)\n    show_rgb_histogram(img, ax2)\n    plt.tight_layout()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"From the sample it looks like histogram of malignant looks a bit more spread out, but we cannot make that conclusion with just few sample. Anyway, it gives us a rough idea about color distribution which might be helpful if we need a specific filter for specific color.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}