{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The Images \n\nThere are 2 types of images containing the same information:\n1. `.dcm` files: [DICOM files](https://en.wikipedia.org/wiki/DICOM). It's saved in the \"Digital Imaging and Communications in Medicine\" format. It contains an image from a medical scan, such as an ultrasound or MRI + information about the patient.\n2. `.jpeg` files: the DICOM files converted into .jpeg format\n3. `.tfrec` files: [The TFRecord file format is a simple record-oriented binary format for ML training data.](https://docs.databricks.com/applications/deep-learning/data-prep/tfrecords-to-tensorflow.html#:~:text=The%20TFRecord%20file%20format%20is,part%20of%20an%20input%20pipeline.)","metadata":{}},{"cell_type":"code","source":"print('Train .dcm number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/train'))), '\\n' +\n      'Test .dcm number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/test'))), '\\n' +\n      'Train .jpeg number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/jpeg/train'))), '\\n' +\n      'Test .jpeg number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/jpeg/test'))), '\\n' +\n      '-----------------------', '\\n' +\n      'There is the same number of images as in train/ test .csv datasets')","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:41:22.491558Z","iopub.execute_input":"2021-10-29T18:41:22.491829Z","iopub.status.idle":"2021-10-29T18:41:25.713358Z","shell.execute_reply.started":"2021-10-29T18:41:22.491801Z","shell.execute_reply":"2021-10-29T18:41:25.712572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Importing libraries**","metadata":{}},{"cell_type":"code","source":"# Regular Imports\nimport os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.image as mpimg\nfrom tabulate import tabulate\nimport missingno as msno \nfrom IPython.display import display_html\nfrom PIL import Image\nimport gc\nimport cv2\n\nimport pydicom # for DICOM images\nfrom skimage.transform import resize\n\n# SKLearn\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Set Color Palettes for the notebook\ncolors_nude = ['#e0798c','#65365a','#da8886','#cfc4c4','#dfd7ca']\nsns.palplot(sns.color_palette(colors_nude))\n\n# Set Style\nsns.set_style(\"whitegrid\")\nsns.despine(left=True, bottom=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:40:55.352968Z","iopub.execute_input":"2021-10-29T18:40:55.353228Z","iopub.status.idle":"2021-10-29T18:40:57.180477Z","shell.execute_reply.started":"2021-10-29T18:40:55.353202Z","shell.execute_reply":"2021-10-29T18:40:57.179649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # Processing","metadata":{}},{"cell_type":"code","source":"list(os.listdir('../input/siim-isic-melanoma-classification'))","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:41:29.303655Z","iopub.execute_input":"2021-10-29T18:41:29.303947Z","iopub.status.idle":"2021-10-29T18:41:29.310765Z","shell.execute_reply.started":"2021-10-29T18:41:29.303917Z","shell.execute_reply":"2021-10-29T18:41:29.310022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Directory\ndirectory = '../input/siim-isic-melanoma-classification'\n\n# Import train and test csv\ntrain_df = pd.read_csv(directory + '/train.csv')\ntest_df = pd.read_csv(directory + '/test.csv')\n\nprint('Train has {:,} rows and Test has {:,} rows.'.format(len(train_df), len(test_df)))\n\n# Change columns names\nnew_names = ['dcm_name', 'ID', 'sex', 'age', 'anatomy', 'diagnosis', 'benign_malignant', 'target']\ntrain_df.columns = new_names\ntest_df.columns = new_names[:5]","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:41:31.48899Z","iopub.execute_input":"2021-10-29T18:41:31.489258Z","iopub.status.idle":"2021-10-29T18:41:31.614626Z","shell.execute_reply.started":"2021-10-29T18:41:31.489229Z","shell.execute_reply":"2021-10-29T18:41:31.613828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head().style.set_caption('Head Train Data')\n","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:41:34.564237Z","iopub.execute_input":"2021-10-29T18:41:34.564679Z","iopub.status.idle":"2021-10-29T18:41:34.651394Z","shell.execute_reply.started":"2021-10-29T18:41:34.56464Z","shell.execute_reply":"2021-10-29T18:41:34.650229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head().style.set_caption('Head Test Data')","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:41:37.782296Z","iopub.execute_input":"2021-10-29T18:41:37.782964Z","iopub.status.idle":"2021-10-29T18:41:37.790826Z","shell.execute_reply.started":"2021-10-29T18:41:37.782927Z","shell.execute_reply":"2021-10-29T18:41:37.789942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Missing Values","metadata":{}},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:41:43.940953Z","iopub.execute_input":"2021-10-29T18:41:43.941242Z","iopub.status.idle":"2021-10-29T18:41:43.975206Z","shell.execute_reply.started":"2021-10-29T18:41:43.941214Z","shell.execute_reply":"2021-10-29T18:41:43.974323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:41:55.142857Z","iopub.execute_input":"2021-10-29T18:41:55.143776Z","iopub.status.idle":"2021-10-29T18:41:55.840565Z","shell.execute_reply.started":"2021-10-29T18:41:55.143723Z","shell.execute_reply":"2021-10-29T18:41:55.839718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1. GENDER COLUMN**","metadata":{}},{"cell_type":"code","source":"nan_sex = train_df[train_df['sex'].isna() == True]\nnan_sex","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:41:59.462595Z","iopub.execute_input":"2021-10-29T18:41:59.463051Z","iopub.status.idle":"2021-10-29T18:41:59.496564Z","shell.execute_reply.started":"2021-10-29T18:41:59.46302Z","shell.execute_reply":"2021-10-29T18:41:59.495684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"is_sex = train_df[train_df['sex'].isna() == False]\nis_sex","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:42:03.783935Z","iopub.execute_input":"2021-10-29T18:42:03.784411Z","iopub.status.idle":"2021-10-29T18:42:03.809881Z","shell.execute_reply.started":"2021-10-29T18:42:03.784365Z","shell.execute_reply":"2021-10-29T18:42:03.809048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Figure\nf, (ax1, ax2) = plt.subplots(1, 2, figsize = (16, 6))\n\na = sns.countplot(nan_sex['anatomy'], ax = ax1, palette=colors_nude)\nb = sns.countplot(is_sex['anatomy'], ax = ax2, palette=colors_nude)\nax1.set_title('NAN Gender: Anatomy', fontsize=16)\nax2.set_title('Rest Gender: Anatomy', fontsize=16)\n\na.set_xticklabels(a.get_xticklabels(), rotation=35, ha=\"right\")\nb.set_xticklabels(b.get_xticklabels(), rotation=35, ha=\"right\")\nsns.despine(left=True, bottom=True);\n\n# Benign/ Malignant check\nprint('Out of 65 NAN values, {} are benign and 0 malignant.'.format(nan_sex['benign_malignant'].value_counts()[0]))","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:42:15.575056Z","iopub.execute_input":"2021-10-29T18:42:15.575713Z","iopub.status.idle":"2021-10-29T18:42:16.089052Z","shell.execute_reply.started":"2021-10-29T18:42:15.575647Z","shell.execute_reply":"2021-10-29T18:42:16.088243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"anatomy = ['lower extremity', 'upper extremity', 'torso']\ntrain_df[(train_df['anatomy'].isin(anatomy)) & (train_df['target'] == 0)]['sex'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:42:25.492662Z","iopub.execute_input":"2021-10-29T18:42:25.49309Z","iopub.status.idle":"2021-10-29T18:42:25.511619Z","shell.execute_reply.started":"2021-10-29T18:42:25.493057Z","shell.execute_reply":"2021-10-29T18:42:25.510767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['sex'].fillna(\"male\", inplace = True) ","metadata":{"execution":{"iopub.status.busy":"2021-10-29T18:42:22.719357Z","iopub.execute_input":"2021-10-29T18:42:22.719641Z","iopub.status.idle":"2021-10-29T18:42:22.72882Z","shell.execute_reply.started":"2021-10-29T18:42:22.719609Z","shell.execute_reply":"2021-10-29T18:42:22.72797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2. AGE COLUMN**","metadata":{}},{"cell_type":"code","source":"nan_age = train_df[train_df['age'].isna() == True]\nnan_age","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:17:25.078458Z","iopub.execute_input":"2021-10-22T18:17:25.079018Z","iopub.status.idle":"2021-10-22T18:17:25.100374Z","shell.execute_reply.started":"2021-10-22T18:17:25.078963Z","shell.execute_reply":"2021-10-22T18:17:25.099475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nis_age = train_df[train_df['age'].isna() == False]\nis_age","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:17:28.029211Z","iopub.execute_input":"2021-10-22T18:17:28.029509Z","iopub.status.idle":"2021-10-22T18:17:28.051984Z","shell.execute_reply.started":"2021-10-22T18:17:28.029473Z","shell.execute_reply":"2021-10-22T18:17:28.050999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Figure\nf, (ax1, ax2) = plt.subplots(1, 2, figsize = (16, 6))\n\na = sns.countplot(nan_age['anatomy'], ax = ax1, palette=colors_nude)\nb = sns.countplot(is_age['anatomy'], ax = ax2, palette=colors_nude)\nax1.set_title('NAN age: Anatomy', fontsize=16)\nax2.set_title('Rest age: Anatomy', fontsize=16)\n\na.set_xticklabels(a.get_xticklabels(), rotation=35, ha=\"right\")\nb.set_xticklabels(b.get_xticklabels(), rotation=35, ha=\"right\")\nsns.despine(left=True, bottom=True);\n\n# Benign/ Malignant check\nprint('Out of 68 NAN values, {} are benign and 0 malignant.'.format(nan_age['benign_malignant'].value_counts()[0]))","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:17:34.555782Z","iopub.execute_input":"2021-10-22T18:17:34.556299Z","iopub.status.idle":"2021-10-22T18:17:35.010508Z","shell.execute_reply.started":"2021-10-22T18:17:34.556265Z","shell.execute_reply":"2021-10-22T18:17:35.009578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the mean age\nanatomy = ['lower extremity', 'upper extremity', 'torso']\nmedian = train_df[(train_df['anatomy'].isin(anatomy)) & (train_df['target'] == 0) & (train_df['sex'] == 'male')]['age'].median()\nprint('Median is:', median)","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:17:38.369456Z","iopub.execute_input":"2021-10-22T18:17:38.370007Z","iopub.status.idle":"2021-10-22T18:17:38.389544Z","shell.execute_reply.started":"2021-10-22T18:17:38.369964Z","shell.execute_reply":"2021-10-22T18:17:38.388885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Impute the missing values with male\ntrain_df['age'].fillna(median, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:17:43.809952Z","iopub.execute_input":"2021-10-22T18:17:43.810511Z","iopub.status.idle":"2021-10-22T18:17:43.816926Z","shell.execute_reply.started":"2021-10-22T18:17:43.810461Z","shell.execute_reply":"2021-10-22T18:17:43.815667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**3. Anaotmy Column**","metadata":{}},{"cell_type":"code","source":"anatomy = train_df.copy()\nanatomy['flag'] = np.where(train_df['anatomy'].isna()==True, 'missing', 'not_missing')","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:18:02.328194Z","iopub.execute_input":"2021-10-22T18:18:02.328906Z","iopub.status.idle":"2021-10-22T18:18:02.344484Z","shell.execute_reply.started":"2021-10-22T18:18:02.32886Z","shell.execute_reply":"2021-10-22T18:18:02.343759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"anatomy = train_df.copy()\nanatomy['flag'] = np.where(train_df['anatomy'].isna()==True, 'missing', 'not_missing')\n\n# Figure\nf, (ax1, ax2) = plt.subplots(1, 2, figsize = (16, 6))\n\nsns.countplot(anatomy['flag'], hue=anatomy['sex'], ax=ax1, palette=colors_nude)\n\nsns.distplot(anatomy[anatomy['flag'] == 'missing']['age'], \n             hist=False, rug=True, label='Missing', ax=ax2, \n             color=colors_nude[2], kde_kws=dict(linewidth=4))\nsns.distplot(anatomy[anatomy['flag'] == 'not_missing']['age'], \n             hist=False, rug=True, label='Not Missing', ax=ax2, \n             color=colors_nude[3], kde_kws=dict(linewidth=4))\n\nax1.set_title('Gender for Anatomy', fontsize=16)\nax2.set_title('Age Distribution for Anatomy', fontsize=16)\nsns.despine(left=True, bottom=True);\n\n# Benign - malignant\nben_mal = anatomy[anatomy['flag'] == 'missing']['benign_malignant'].value_counts()\nprint('From all missing values, {} are benign and {} malignant.'.format(ben_mal[0], ben_mal[1]))","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:18:20.615997Z","iopub.execute_input":"2021-10-22T18:18:20.616524Z","iopub.status.idle":"2021-10-22T18:18:22.073882Z","shell.execute_reply.started":"2021-10-22T18:18:20.616491Z","shell.execute_reply":"2021-10-22T18:18:22.072928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Impute for anatomy\ntrain_df['anatomy'].fillna('torso', inplace = True) ","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:19:37.427954Z","iopub.execute_input":"2021-10-22T18:19:37.428254Z","iopub.status.idle":"2021-10-22T18:19:37.437934Z","shell.execute_reply.started":"2021-10-22T18:19:37.428221Z","shell.execute_reply":"2021-10-22T18:19:37.436947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Target Variable:\n1. Very HIGH class imbalance. We need to take this in consideration when Modeling.\n2. Age distribution:\n    * Benign: follows a normal distribution\n    * Malignant: a little skewed to the left, with the peak oriented towards higher age values.","metadata":{}},{"cell_type":"code","source":"# Figure\nf, (ax1, ax2) = plt.subplots(1, 2, figsize = (16, 6))\n\na = sns.countplot(data = train_df, x = 'benign_malignant', palette=colors_nude[2:4],\n                 ax=ax1)\nb = sns.distplot(a = train_df[train_df['target']==0]['age'], ax=ax2, color=colors_nude[2], \n                 hist=False, rug=True, kde_kws=dict(linewidth=4), label='Benign')\nc = sns.distplot(a = train_df[train_df['target']==1]['age'], ax=ax2, color=colors_nude[3], \n                 hist=False, rug=True, kde_kws=dict(linewidth=4), label='Malignant')\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n    \nax1.set_title('Frequency for Target Variable', fontsize=16)\nax2.set_title('Age Distribution the Target types', fontsize=16)\nsns.despine(left=True, bottom=True);","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:19:44.492574Z","iopub.execute_input":"2021-10-22T18:19:44.492877Z","iopub.status.idle":"2021-10-22T18:19:45.744078Z","shell.execute_reply.started":"2021-10-22T18:19:44.492844Z","shell.execute_reply":"2021-10-22T18:19:45.743234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Relationship between target variable and gender column**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\na = sns.countplot(data=train_df, x='benign_malignant', hue='sex', palette=colors_nude)\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n\nplt.title('Gender split by Target Variable', fontsize=16)\nsns.despine(left=True, bottom=True);","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:19:49.377164Z","iopub.execute_input":"2021-10-22T18:19:49.377445Z","iopub.status.idle":"2021-10-22T18:19:49.689904Z","shell.execute_reply.started":"2021-10-22T18:19:49.377417Z","shell.execute_reply":"2021-10-22T18:19:49.688915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. There are more males than females in the dataset\n2. However, the percentages are ~ the same","metadata":{}},{"cell_type":"markdown","source":"**Relationship between target and anatomy**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\na = sns.countplot(data=train_df, x='benign_malignant', hue='anatomy', palette=colors_nude)\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n\nplt.title('Anatomy split by Target Variable', fontsize=16)\nsns.despine(left=True, bottom=True);","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:19:55.556417Z","iopub.execute_input":"2021-10-22T18:19:55.556736Z","iopub.status.idle":"2021-10-22T18:19:56.050235Z","shell.execute_reply.started":"2021-10-22T18:19:55.556696Z","shell.execute_reply":"2021-10-22T18:19:56.049257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the files\ntrain_df.to_csv('train_clean.csv', index=False)\ntest_df.to_csv('test_clean.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:19:59.736046Z","iopub.execute_input":"2021-10-22T18:19:59.736327Z","iopub.status.idle":"2021-10-22T18:19:59.973914Z","shell.execute_reply.started":"2021-10-22T18:19:59.736299Z","shell.execute_reply":"2021-10-22T18:19:59.972969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Add image path**","metadata":{}},{"cell_type":"code","source":"# === DICOM ===\n# Create the paths\npath_train = directory + '/train/' + train_df['dcm_name'] + '.dcm'\npath_test = directory + '/test/' + test_df['dcm_name'] + '.dcm'\n\n# Append to the original dataframes\ntrain_df['path_dicom'] = path_train\ntest_df['path_dicom'] = path_test\n\n# === JPEG ===\n# Create the paths\npath_train = directory + '/jpeg/train/' + train_df['dcm_name'] + '.jpg'\npath_test = directory + '/jpeg/test/' + test_df['dcm_name'] + '.jpg'\n\n# Append to the original dataframes\ntrain_df['path_jpeg'] = path_train\ntest_df['path_jpeg'] = path_test","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:20:07.122084Z","iopub.execute_input":"2021-10-22T18:20:07.12236Z","iopub.status.idle":"2021-10-22T18:20:07.164762Z","shell.execute_reply.started":"2021-10-22T18:20:07.122331Z","shell.execute_reply":"2021-10-22T18:20:07.163876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Performing one hot encoding**","metadata":{}},{"cell_type":"code","source":"# === TRAIN ===\nto_encode = ['sex', 'anatomy', 'diagnosis']\nencoded_all = []\n\nlabel_encoder = LabelEncoder()\n\nfor column in to_encode:\n    encoded = label_encoder.fit_transform(train_df[column])\n    encoded_all.append(encoded)\n    \ntrain_df['sex'] = encoded_all[0]\ntrain_df['anatomy'] = encoded_all[1]\ntrain_df['diagnosis'] = encoded_all[2]\n\nif 'benign_malignant' in train_df.columns : train_df.drop(['benign_malignant'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:20:11.811638Z","iopub.execute_input":"2021-10-22T18:20:11.811955Z","iopub.status.idle":"2021-10-22T18:20:11.871125Z","shell.execute_reply.started":"2021-10-22T18:20:11.811919Z","shell.execute_reply":"2021-10-22T18:20:11.870512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nTransforming all categorical features un numerical.\n> Note1: `sex`, `anatomy`, `diagnosis` need to be encoded.\n\n> Note2: `benign_malignant` column will be dropped, as the information is already in the `target` column.","metadata":{}},{"cell_type":"markdown","source":"# Folds","metadata":{}},{"cell_type":"code","source":"shapes_train = []\n\nfor k, path in enumerate(train_df['path_jpeg']):\n    image = Image.open(path)\n    shapes_train.append(image.size)\n    \n    if k >= 100: break\n        \nshapes_train = pd.DataFrame(data = shapes_train, columns = ['H', 'W'], dtype='object')\nshapes_train['Size'] = '[' + shapes_train['H'].astype(str) + ', ' + shapes_train['W'].astype(str) + ']'","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:20:17.638564Z","iopub.execute_input":"2021-10-22T18:20:17.639252Z","iopub.status.idle":"2021-10-22T18:20:20.563646Z","shell.execute_reply.started":"2021-10-22T18:20:17.639215Z","shell.execute_reply":"2021-10-22T18:20:20.562733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (16, 6))\n\na = sns.countplot(shapes_train['Size'], palette=colors_nude)\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n    \nplt.title('100 Images Shapes', fontsize=16)\nsns.despine(left=True, bottom=True);","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:20:20.748099Z","iopub.execute_input":"2021-10-22T18:20:20.74841Z","iopub.status.idle":"2021-10-22T18:20:21.091903Z","shell.execute_reply.started":"2021-10-22T18:20:20.748375Z","shell.execute_reply":"2021-10-22T18:20:21.091018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Benign vs Malignant**","metadata":{}},{"cell_type":"code","source":"def show_images(data, n = 5, rows=1, cols=5, title='Default'):\n    plt.figure(figsize=(16,4))\n\n    for k, path in enumerate(data['path_dicom'][:n]):\n        image = pydicom.read_file(path)\n        image = image.pixel_array\n        \n        # image = resize(image, (200, 200), anti_aliasing=True)\n\n        plt.suptitle(title, fontsize = 16)\n        plt.subplot(rows, cols, k+1)\n        plt.imshow(image)\n        plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:20:31.667951Z","iopub.execute_input":"2021-10-22T18:20:31.668215Z","iopub.status.idle":"2021-10-22T18:20:31.675236Z","shell.execute_reply.started":"2021-10-22T18:20:31.668187Z","shell.execute_reply":"2021-10-22T18:20:31.674224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show Benign Samples\nshow_images(train_df[train_df['target'] == 0], n=10, rows=2, cols=5, title='Benign Sample')","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:20:44.060646Z","iopub.execute_input":"2021-10-22T18:20:44.060934Z","iopub.status.idle":"2021-10-22T18:21:04.724756Z","shell.execute_reply.started":"2021-10-22T18:20:44.060905Z","shell.execute_reply":"2021-10-22T18:21:04.723756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show Malignant Samples\nshow_images(train_df[train_df['target'] == 1], n=10, rows=2, cols=5, title='Malignant Sample')","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:21:04.726493Z","iopub.execute_input":"2021-10-22T18:21:04.727245Z","iopub.status.idle":"2021-10-22T18:21:22.053006Z","shell.execute_reply.started":"2021-10-22T18:21:04.727195Z","shell.execute_reply":"2021-10-22T18:21:22.052386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # Augmentations","metadata":{}},{"cell_type":"markdown","source":"**B& W view**","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=6, figsize=(16,6))\nplt.suptitle(\"B&W\", fontsize = 16)\n\nfor i in range(0, 2*6):\n    data = pydicom.read_file(train_df['path_dicom'][i])\n    image = data.pixel_array\n    \n    # Transform to B&W\n    # The function converts an input image from one color space to another.\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n    image = cv2.resize(image, (200,200))\n    \n    x = i // 6\n    y = i % 6\n    axes[x, y].imshow(image, cmap=plt.cm.bone) \n    axes[x, y].axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:21:22.053934Z","iopub.execute_input":"2021-10-22T18:21:22.054582Z","iopub.status.idle":"2021-10-22T18:21:28.19342Z","shell.execute_reply.started":"2021-10-22T18:21:22.054548Z","shell.execute_reply":"2021-10-22T18:21:28.192666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=6, figsize=(16,6))\nplt.suptitle(\"Hue, Saturation, Brightness\", fontsize = 16)\n\nfor i in range(0, 2*6):\n    data = pydicom.read_file(train_df['path_dicom'][i])\n    image = data.pixel_array\n    \n    # Transform to B&W\n    # The function converts an input image from one color space to another.\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2HLS)\n    image = cv2.resize(image, (200,200))\n    \n    x = i // 6\n    y = i % 6\n    axes[x, y].imshow(image, cmap=plt.cm.bone) \n    axes[x, y].axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:21:28.195309Z","iopub.execute_input":"2021-10-22T18:21:28.196084Z","iopub.status.idle":"2021-10-22T18:21:34.422888Z","shell.execute_reply.started":"2021-10-22T18:21:28.196042Z","shell.execute_reply":"2021-10-22T18:21:34.422219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Importing torch library for augmentations**","metadata":{}},{"cell_type":"code","source":"# Necessary Imports\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nimport torchvision.transforms as transforms\nimport torchvision\n# Select a small sample of the .jpeg image paths\nimage_list = train_df.sample(12)['path_jpeg']\nimage_list = image_list.reset_index()['path_jpeg']\n\n# Show the sample\nplt.figure(figsize=(16,6))\nplt.suptitle(\"Original View\", fontsize = 16)\n    \nfor k, path in enumerate(image_list):\n    image = mpimg.imread(path)\n        \n    plt.subplot(2, 6, k+1)\n    plt.imshow(image)\n    plt.axis('off')\n\n    ","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:22:35.035237Z","iopub.execute_input":"2021-10-22T18:22:35.035473Z","iopub.status.idle":"2021-10-22T18:22:58.976614Z","shell.execute_reply.started":"2021-10-22T18:22:35.035443Z","shell.execute_reply":"2021-10-22T18:22:58.975733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DatasetExample(Dataset):\n    def __init__(self, image_list, transforms=None):\n        self.image_list = image_list\n        self.transforms = transforms\n    \n    # To get item's length\n    def __len__(self):\n        return (len(self.image_list))\n    \n    # For indexing\n    def __getitem__(self, i):\n        # Read in image\n        image = plt.imread(self.image_list[i])\n        image = Image.fromarray(image).convert('RGB')        \n        image = np.asarray(image).astype(np.uint8)\n        if self.transforms is not None:\n            image = self.transforms(image)\n            \n        return torch.tensor(image, dtype=torch.float)","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:22:58.977851Z","iopub.execute_input":"2021-10-22T18:22:58.978094Z","iopub.status.idle":"2021-10-22T18:22:58.985835Z","shell.execute_reply.started":"2021-10-22T18:22:58.978057Z","shell.execute_reply":"2021-10-22T18:22:58.984964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predefined Show Images Function\ndef show_transform(image, title=\"Default\"):\n    plt.figure(figsize=(16,6))\n    plt.suptitle(title, fontsize = 16)\n    \n    # Unnormalize\n    image = image / 2 + 0.5  \n    npimg = image.numpy()\n    npimg = np.clip(npimg, 0., 1.)\n    plt.imshow(np.transpose(npimg, (1, 2, 0)))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:22:58.987109Z","iopub.execute_input":"2021-10-22T18:22:58.987526Z","iopub.status.idle":"2021-10-22T18:22:59.002407Z","shell.execute_reply.started":"2021-10-22T18:22:58.987481Z","shell.execute_reply":"2021-10-22T18:22:59.001395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1.Crop**","metadata":{}},{"cell_type":"code","source":"# Transform\ntransform = transforms.Compose([\n     transforms.ToPILImage(),\n     transforms.Resize((300, 300)),\n     transforms.CenterCrop((100, 100)),\n     transforms.ToTensor(),\n     transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5)),\n     ])\n\n# Create the dataset\npytorch_dataset = DatasetExample(image_list=image_list, transforms=transform)\npytorch_dataloader = DataLoader(dataset=pytorch_dataset, batch_size=12, shuffle=True)\n\n# Select the data\nimages = next(iter(pytorch_dataloader))\n \n# show images\nshow_transform(torchvision.utils.make_grid(images, nrow=6), title=\"Crop\")","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:22:06.440608Z","iopub.execute_input":"2021-10-22T18:22:06.441008Z","iopub.status.idle":"2021-10-22T18:22:18.061053Z","shell.execute_reply.started":"2021-10-22T18:22:06.44094Z","shell.execute_reply":"2021-10-22T18:22:18.059852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2.Color jitter**","metadata":{}},{"cell_type":"code","source":"#  Transform\ntransform = transforms.Compose([\n     transforms.ToPILImage(),\n     transforms.Resize((300, 300)),\n     transforms.ColorJitter(brightness=0.7, contrast=0.7, saturation=0.7, hue=0.5),\n     transforms.ToTensor(),\n     transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5)),\n     ])\n\n# Create the dataset\npytorch_dataset = DatasetExample(image_list=image_list, transforms=transform)\npytorch_dataloader = DataLoader(dataset=pytorch_dataset, batch_size=12, shuffle=True)\n\n# Select the data\nimages = next(iter(pytorch_dataloader))\n \n# show images\nshow_transform(torchvision.utils.make_grid(images, nrow=6), title=\"Color Jitter\")","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:22:59.003972Z","iopub.execute_input":"2021-10-22T18:22:59.00424Z","iopub.status.idle":"2021-10-22T18:23:06.281049Z","shell.execute_reply.started":"2021-10-22T18:22:59.00421Z","shell.execute_reply":"2021-10-22T18:23:06.280195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Random Grey Scale**","metadata":{}},{"cell_type":"code","source":"# Transform\ntransform = transforms.Compose([\n     transforms.ToPILImage(),\n     transforms.Resize((300, 300)),\n     transforms.RandomGrayscale(p=0.7),\n     transforms.ToTensor(),\n     transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5)),\n     ])\n\n# Create the dataset\npytorch_dataset = DatasetExample(image_list=image_list, transforms=transform)\npytorch_dataloader = DataLoader(dataset=pytorch_dataset, batch_size=12, shuffle=True)\n\n# Select the data\nimages = next(iter(pytorch_dataloader))\n \n# show images\nshow_transform(torchvision.utils.make_grid(images, nrow=6), title=\"Random Greyscale\")","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:22:18.062935Z","iopub.execute_input":"2021-10-22T18:22:18.063244Z","iopub.status.idle":"2021-10-22T18:22:26.669551Z","shell.execute_reply.started":"2021-10-22T18:22:18.063208Z","shell.execute_reply":"2021-10-22T18:22:26.668741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Random Vertical flip**","metadata":{}},{"cell_type":"code","source":"# Transform\ntransform = transforms.Compose([\n     transforms.ToPILImage(),\n     transforms.Resize((300, 300)),\n     transforms.RandomVerticalFlip(p=0.7),\n     transforms.ToTensor(),\n     transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5)),\n     ])\n\n# Create the dataset\npytorch_dataset = DatasetExample(image_list=image_list, transforms=transform)\npytorch_dataloader = DataLoader(dataset=pytorch_dataset, batch_size=12, shuffle=True)\n\n# Select the data\nimages = next(iter(pytorch_dataloader))\n \n# show images\nshow_transform(torchvision.utils.make_grid(images, nrow=6), title=\"Random Vertical Flip\")","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:22:26.67088Z","iopub.execute_input":"2021-10-22T18:22:26.671564Z","iopub.status.idle":"2021-10-22T18:22:35.032933Z","shell.execute_reply.started":"2021-10-22T18:22:26.671527Z","shell.execute_reply":"2021-10-22T18:22:35.032024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Hair Removal**","metadata":{}},{"cell_type":"code","source":"def hair_remove(image):\n    # convert image to grayScale\n    grayScale = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n\n    # kernel for morphologyEx\n    kernel = cv2.getStructuringElement(1,(17,17))\n\n    # apply MORPH_BLACKHAT to grayScale image\n    blackhat = cv2.morphologyEx(grayScale, cv2.MORPH_BLACKHAT, kernel)\n\n    # apply thresholding to blackhat\n    _,threshold = cv2.threshold(blackhat,10,255,cv2.THRESH_BINARY)\n\n    # inpaint with original image and threshold image\n    final_image = cv2.inpaint(image,threshold,1,cv2.INPAINT_TELEA)\n\n    return final_image","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:23:06.282228Z","iopub.execute_input":"2021-10-22T18:23:06.282437Z","iopub.status.idle":"2021-10-22T18:23:06.288556Z","shell.execute_reply.started":"2021-10-22T18:23:06.282411Z","shell.execute_reply":"2021-10-22T18:23:06.287946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select a small sample of the .jpeg image paths\n# We select some hairy photos on purpose\nhairy_photos = train_df[train_df[\"sex\"] == 1].reset_index().iloc[[12, 14, 17, 22, 33, 34]]\nimage_list = hairy_photos['path_jpeg']\nimage_list = image_list.reset_index()['path_jpeg']","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:23:06.290815Z","iopub.execute_input":"2021-10-22T18:23:06.291044Z","iopub.status.idle":"2021-10-22T18:23:06.312978Z","shell.execute_reply.started":"2021-10-22T18:23:06.291016Z","shell.execute_reply":"2021-10-22T18:23:06.312316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the Sample Images\nplt.figure(figsize=(16,3))\nplt.suptitle(\"Original Hairy Images\", fontsize = 16)\n    \nfor k, path in enumerate(image_list):\n    image = mpimg.imread(path)\n    image = cv2.resize(image,(300, 300))\n        \n    plt.subplot(1, 6, k+1)\n    plt.imshow(image)\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:23:06.314407Z","iopub.execute_input":"2021-10-22T18:23:06.314891Z","iopub.status.idle":"2021-10-22T18:23:09.191933Z","shell.execute_reply.started":"2021-10-22T18:23:06.314859Z","shell.execute_reply":"2021-10-22T18:23:09.190822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the Augmented\nplt.figure(figsize=(16,3))\nplt.suptitle(\"Non Hairy Images\", fontsize = 16)\n    \nfor k, path in enumerate(image_list):\n    image = mpimg.imread(path)\n    image = cv2.resize(image,(300, 300))\n    image = hair_remove(image)\n        \n    plt.subplot(1, 6, k+1)\n    plt.imshow(image)\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-10-22T18:23:09.193317Z","iopub.execute_input":"2021-10-22T18:23:09.193613Z","iopub.status.idle":"2021-10-22T18:23:13.673337Z","shell.execute_reply.started":"2021-10-22T18:23:09.193577Z","shell.execute_reply":"2021-10-22T18:23:13.672468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}