{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"deb1aee0-8361-4bef-91b9-a343b0ef3240","_cell_guid":"41cfb4db-de24-4a75-96d7-e069045ee0b4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T08:04:01.245555Z","iopub.execute_input":"2022-08-11T08:04:01.245979Z","iopub.status.idle":"2022-08-11T08:04:01.252502Z","shell.execute_reply.started":"2022-08-11T08:04:01.245946Z","shell.execute_reply":"2022-08-11T08:04:01.250900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Regular imports\nimport os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom tabulate import tabulate\nimport missingno as msno #need to see\nfrom IPython.display import display_html\nfrom PIL import Image\nimport gc #need to see\nimport cv2\nfrom sklearn.preprocessing import LabelEncoder\n\nimport pydicom #for DICOM images\nfrom skimage.transform import resize\n\nimport warnings #need to see\nwarnings.filterwarnings(\"ignore\")\n\n#set colour palettes for the notebook\ncolors_nude = ['#e0798c','#65365a','#da8886','#cfc4c4','#dfd7ca']\nsns.palplot(sns.color_palette(colors_nude))\n\n#set style\nsns.set_style(\"whitegrid\")\nsns.despine(left = True, bottom = True) # need to see","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:01.274224Z","iopub.execute_input":"2022-08-11T08:04:01.275178Z","iopub.status.idle":"2022-08-11T08:04:02.661667Z","shell.execute_reply.started":"2022-08-11T08:04:01.275133Z","shell.execute_reply":"2022-08-11T08:04:02.659417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(sorted(os.listdir(\"../input/siim-isic-melanoma-classification\")))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:02.665197Z","iopub.execute_input":"2022-08-11T08:04:02.665832Z","iopub.status.idle":"2022-08-11T08:04:02.681528Z","shell.execute_reply.started":"2022-08-11T08:04:02.665772Z","shell.execute_reply":"2022-08-11T08:04:02.679354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3.CSV Files_Train & Test","metadata":{}},{"cell_type":"code","source":"#directory\ndirectory = \"../input/siim-isic-melanoma-classification\"\n#import the 2 csv \ntrain_df = pd.read_csv(directory + \"/train.csv\")\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:02.686838Z","iopub.execute_input":"2022-08-11T08:04:02.688257Z","iopub.status.idle":"2022-08-11T08:04:02.814979Z","shell.execute_reply.started":"2022-08-11T08:04:02.688187Z","shell.execute_reply":"2022-08-11T08:04:02.813505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(directory + \"/test.csv\")\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:02.818261Z","iopub.execute_input":"2022-08-11T08:04:02.819056Z","iopub.status.idle":"2022-08-11T08:04:02.861683Z","shell.execute_reply.started":"2022-08-11T08:04:02.819007Z","shell.execute_reply":"2022-08-11T08:04:02.859954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train has {:,} rows and Test has {:,} rows.\".format(len(train_df),len(test_df)))\n\n#change column names\nnew_names = [\"dcm_name\",\"ID\",\"sex\",\"age\",\"anatomy\",\"diagnosis\",\"benign_malignant\",\"target\"]\n#new_names[0:5]\ntrain_df.columns = new_names\ntest_df.columns = new_names[:5]\ntest_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:02.863508Z","iopub.execute_input":"2022-08-11T08:04:02.863896Z","iopub.status.idle":"2022-08-11T08:04:02.878310Z","shell.execute_reply.started":"2022-08-11T08:04:02.863855Z","shell.execute_reply":"2022-08-11T08:04:02.877467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f,(ax1,ax2) = plt.subplots(1,2,figsize =(16,6))\nmsno.matrix(train_df,ax = ax1,color = (207/255,196/255,171/255),fontsize = 10)\nmsno.matrix(test_df, ax = ax2, color = (218/255,136/255, 130/255), fontsize = 10)\n\nax1.set_title(\"Train Missing Values Map\", fontsize = 16)\nax2.set_title(\"Test Missing Values Map\" , fontsize = 16)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:02.879558Z","iopub.execute_input":"2022-08-11T08:04:02.880445Z","iopub.status.idle":"2022-08-11T08:04:03.458120Z","shell.execute_reply.started":"2022-08-11T08:04:02.880409Z","shell.execute_reply":"2022-08-11T08:04:03.456355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_sex = train_df[train_df['sex'].isna() == True]\nprint(f\"Number of NaN values in Sex Column is :{nan_sex.shape[0]}\")\nis_sex = train_df[train_df['sex'].isna()==False]\nprint(f\"Number of Existing values in Sex Column is: {is_sex.shape[0]}\")\n#figure \nf, (ax1, ax2) = plt.subplots(1,2,figsize = (16,6))\n\na = sns.countplot(nan_sex[\"anatomy\"], ax = ax1, palette = colors_nude)\nb = sns.countplot(is_sex[\"anatomy\"], ax = ax2, palette = colors_nude)\nax1.set_title(\"NAN Gender: Anatomy\", fontsize = 16)\nax2.set_title(\"Rest Gender:Anatomy\", fontsize = 16)\n\na.set_xticklabels(a.get_xticklabels(), rotation = 35,ha = \"right\")\nb.set_xticklabels(b.get_xticklabels(), rotation = 35, ha = \"right\")\n\nsns.despine(left = True, bottom = True)\n\nno_benign = nan_sex[\"benign_malignant\"].value_counts()[0]\n#Benign and malignant check\nprint(f\"Out of 65 NaN values,{no_benign} are benign and 0 malignant\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:03.460917Z","iopub.execute_input":"2022-08-11T08:04:03.461824Z","iopub.status.idle":"2022-08-11T08:04:03.821954Z","shell.execute_reply.started":"2022-08-11T08:04:03.461775Z","shell.execute_reply":"2022-08-11T08:04:03.821003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check how many are makes and how many are females\nanatomy = [\"lower extremity\",\"upper extremity\",\"torso\"]\ntrain_df[train_df[\"anatomy\"].isin(anatomy) & (train_df[\"target\"]==0)][\"sex\"].value_counts()\n\n#impute the missing values with male\ntrain_df[\"sex\"].fillna(\"male\", inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:03.823780Z","iopub.execute_input":"2022-08-11T08:04:03.824611Z","iopub.status.idle":"2022-08-11T08:04:03.843932Z","shell.execute_reply.started":"2022-08-11T08:04:03.824562Z","shell.execute_reply":"2022-08-11T08:04:03.843043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train: AGE Variable","metadata":{}},{"cell_type":"code","source":"nan_age = train_df[train_df[\"age\"].isna()== True]\nis_age = train_df[train_df[\"age\"].isna() == False]\n\n#Figure\nf,(ax1,ax2) = plt.subplots(1,2,figsize = (16,6))\n\na = sns.countplot(nan_age[\"anatomy\"], ax = ax1, palette = colors_nude)\nb = sns.countplot(is_age[\"anatomy\"], ax = ax2, palette = colors_nude)\nax1.set_title(\"NAN Age: Anatomy\", fontsize = 16)\nax2.set_title(\"Rest Age:Anatomy\", fontsize = 16)\n\na.set_xticklabels(a.get_xticklabels(), rotation = 35,ha = \"right\")\nb.set_xticklabels(b.get_xticklabels(), rotation = 35, ha = \"right\")\n\nsns.despine(left = True, bottom = True)\n\nno_benign = nan_age[\"benign_malignant\"].value_counts()[0]\n#Benign and malignant check\nprint(f\"Out of 65 NaN values,{no_benign} are benign and 0 malignant\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:03.845216Z","iopub.execute_input":"2022-08-11T08:04:03.846436Z","iopub.status.idle":"2022-08-11T08:04:04.212578Z","shell.execute_reply.started":"2022-08-11T08:04:03.846396Z","shell.execute_reply":"2022-08-11T08:04:04.211173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the mean age\nanatomy = [\"lower extremity\", \"upper extremity\", \"torso\"]\n\nmedian = train_df[(train_df[\"anatomy\"].isin(anatomy)) & (train_df[\"target\"]==0) & (train_df[\"sex\"] == \"male\")][\"age\"].median()\nprint(\"Median is: \" , median)\n\ntrain_df['age'].fillna(median, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:04.216346Z","iopub.execute_input":"2022-08-11T08:04:04.216683Z","iopub.status.idle":"2022-08-11T08:04:04.237617Z","shell.execute_reply.started":"2022-08-11T08:04:04.216653Z","shell.execute_reply":"2022-08-11T08:04:04.236407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train: Anatomy Variable","metadata":{}},{"cell_type":"code","source":"anatomy = train_df.copy()\nanatomy['flag'] = np.where(train_df['anatomy'].isna()== True, \"missing\",\"not_missing\")\n\n#figure\nf, (ax1, ax2) = plt.subplots(1,2, figsize = (16,6))\nsns.countplot(anatomy['flag'], hue = anatomy['sex'], ax = ax1 , palette = colors_nude)\n\n\nsns.distplot(anatomy[anatomy['flag']== 'missing']['age'], hist = False,rug=True,label=\"Missing\",ax = ax2, color = colors_nude[2],kde_kws=dict(linewidth=4))\n\nsns.distplot(anatomy[anatomy['flag']== 'not missing']['age'], hist = False,rug=True,label=\"Missing\",ax = ax2, color = colors_nude[2],kde_kws=dict(linewidth=4))\n\nax1.set_title(\"Gender for Anatomy\", fontsize = 16)\nax2.set_title(\"Age Distribution for Anatomy\", fontsize = 16)\nsns.despine(left = True, bottom = True)\n#benign-malignant\n\nben_malignant = anatomy[anatomy['flag'] == 'missing']['benign_malignant'].value_counts()\nprint(\"From all missing values,{} are benign and {} are malignant\".format(ben_malignant[0],ben_malignant[1]))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:04.239386Z","iopub.execute_input":"2022-08-11T08:04:04.240115Z","iopub.status.idle":"2022-08-11T08:04:04.803243Z","shell.execute_reply.started":"2022-08-11T08:04:04.240068Z","shell.execute_reply":"2022-08-11T08:04:04.801730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#impute for anatomy\ntrain_df['anatomy'].fillna(\"torso\", inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:04.804982Z","iopub.execute_input":"2022-08-11T08:04:04.805359Z","iopub.status.idle":"2022-08-11T08:04:04.814547Z","shell.execute_reply.started":"2022-08-11T08:04:04.805302Z","shell.execute_reply":"2022-08-11T08:04:04.812609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save the files \ntrain_df.to_csv(\"train_clean.csv\" , index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:04.816703Z","iopub.execute_input":"2022-08-11T08:04:04.817857Z","iopub.status.idle":"2022-08-11T08:04:04.939436Z","shell.execute_reply.started":"2022-08-11T08:04:04.817816Z","shell.execute_reply":"2022-08-11T08:04:04.938337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA : Target Variable","metadata":{}},{"cell_type":"code","source":"#figure\nf,(ax1,ax2) = plt.subplots(1,2,figsize =(16,6))\na = sns.countplot(data = train_df, x = 'benign_malignant', palette = colors_nude[2:4], ax = ax1)\nb = sns.distplot(a = train_df[train_df['target']==0]['age'], ax= ax2, color = colors_nude[2],hist = False,rug = True,kde_kws = dict(linewidth =4), label=\"Benign\")\nc = sns.distplot(a = train_df[train_df['target']== 1]['age'], ax= ax2, color = colors_nude[3],hist = False,rug = True,kde_kws = dict(linewidth =4), label=\"Malignant\")\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(),','),\n              (p.get_x() + p.get_width()/2 ,\n              p.get_height()), ha = 'center', va = 'center',\n              xytext = (0,4), textcoords = 'offset points')\n\n\nax1.set_title(\"Frequency for Target Variable\", fontsize = 16)\nax2.set_title(\"Age Distribution for Target Types\", fontsize = 16)\n\nsns.despine(left = True,bottom = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:04.940938Z","iopub.execute_input":"2022-08-11T08:04:04.942145Z","iopub.status.idle":"2022-08-11T08:04:06.222720Z","shell.execute_reply.started":"2022-08-11T08:04:04.942071Z","shell.execute_reply":"2022-08-11T08:04:06.221419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target and Genders","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize =(16,6))\na = sns.countplot(data = train_df, x = \"benign_malignant\",hue = \"sex\",palette = colors_nude)\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','),\n              (p.get_x() + p.get_width()/2,\n               p.get_height()), ha = 'center',va = 'center',\n               xytext = (0,4), textcoords = 'offset points')\n\nplt.title('Gender split by Target Variable', fontsize = 16)\nsns.despine(left = True, bottom = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:06.224292Z","iopub.execute_input":"2022-08-11T08:04:06.224685Z","iopub.status.idle":"2022-08-11T08:04:06.524503Z","shell.execute_reply.started":"2022-08-11T08:04:06.224652Z","shell.execute_reply":"2022-08-11T08:04:06.523167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Anatomy and Diagnosis","metadata":{}},{"cell_type":"code","source":"#figure\nf, (ax1,ax2) = plt.subplots(1,2,figsize = (16,6))\n\na = sns.countplot(train_df['anatomy'], ax = ax1, palette = colors_nude)\nb = sns.countplot(train_df['diagnosis'], ax = ax2, palette = colors_nude)\n\na.set_xticklabels(a.get_xticklabels(), rotation = 35, ha =\"right\")\nb.set_xticklabels(b.get_xticklabels(), rotation = 35, ha = \"right\")\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','),\n              (p.get_x() + p.get_width()/2,\n               p.get_height()), ha = 'center',va = 'center',\n               xytext = (0,4), textcoords = 'offset points')\n\nfor p in b.patches:\n    b.annotate(format(p.get_height(), ','),\n              (p.get_x() + p.get_width()/2,\n               p.get_height()), ha = 'center',va = 'center',\n               xytext = (0,4), textcoords = 'offset points')\n    \nax1.set_title(\"Anatomy Frequencies\" , fontsize = 16)\nax2.set_title(\"Diagnosis Frequencies\", fontsize = 16)\n\nsns.despine(left = True, bottom = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:06.526302Z","iopub.execute_input":"2022-08-11T08:04:06.526762Z","iopub.status.idle":"2022-08-11T08:04:07.099170Z","shell.execute_reply.started":"2022-08-11T08:04:06.526719Z","shell.execute_reply":"2022-08-11T08:04:07.098075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Anatomy and Target","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (16,6))\na = sns.countplot(data = train_df, x = \"benign_malignant\", hue = \"anatomy\", palette = colors_nude)\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','),\n              (p.get_x() + p.get_width()/2,\n               p.get_height()), ha = 'center',va = 'center',\n               xytext = (0,4), textcoords = 'offset points')\n    \nplt.title(\"Anatomy split by Target Variable\", fontsize = 16)\nsns.despine(left = True, bottom = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:07.100770Z","iopub.execute_input":"2022-08-11T08:04:07.101353Z","iopub.status.idle":"2022-08-11T08:04:07.506181Z","shell.execute_reply.started":"2022-08-11T08:04:07.101300Z","shell.execute_reply":"2022-08-11T08:04:07.504772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Diagnosis and Target","metadata":{}},{"cell_type":"code","source":"f, (ax1,ax2)= plt.subplots(1,2,figsize = (16,6))\n\na = sns.countplot(train_df[train_df['target'] == 0]['diagnosis'], ax = ax1,palette = colors_nude)\nb = sns.countplot(train_df[train_df['target']==1]['diagnosis'], ax = ax2, palette = colors_nude)\n\na.set_xticklabels(a.get_xticklabels(), rotation = 35, ha = \"right\")\nb.set_xticklabels(b.get_xticklabels(), rotation = 35, ha = \"right\")\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','),\n              (p.get_x() + p.get_width()/2,\n               p.get_height()), ha = 'center',va = 'center',\n               xytext = (0,4), textcoords = 'offset points')\n    \nfor p in b.patches:\n    b.annotate(format(p.get_height(), ','),\n              (p.get_x() + p.get_width()/2,\n               p.get_height()), ha = 'center',va = 'center',\n               xytext = (0,4), textcoords = 'offset points')\n\nax1.set_title(\"Benign cases:Diagnosis View\" , fontsize = 16)\nax2.set_title(\"Malignant cases: Diagnosis View\" , fontsize = 16)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:07.508397Z","iopub.execute_input":"2022-08-11T08:04:07.508907Z","iopub.status.idle":"2022-08-11T08:04:08.006368Z","shell.execute_reply.started":"2022-08-11T08:04:07.508862Z","shell.execute_reply":"2022-08-11T08:04:08.004167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Patients","metadata":{}},{"cell_type":"code","source":"#Count the number of images per ID\npatients_count_train = train_df.groupby(by =\"ID\")['dcm_name']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:08.008646Z","iopub.execute_input":"2022-08-11T08:04:08.009202Z","iopub.status.idle":"2022-08-11T08:04:08.016426Z","shell.execute_reply.started":"2022-08-11T08:04:08.009142Z","shell.execute_reply":"2022-08-11T08:04:08.015295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients_count_train = train_df.groupby(by =\"ID\")['dcm_name'].count().reset_index()\npatients_count_test = test_df.groupby(by =\"ID\")['dcm_name'].count().reset_index()\n\nf,(ax1,ax2) = plt.subplots(1,2,figsize =(16,6))\n\na = sns.distplot( patients_count_train['dcm_name'],kde = False,bins= 50,\n                 ax = ax1,color = colors_nude[0],hist_kws = {'alpha':1})\n\nb = sns.distplot(patients_count_test['dcm_name'], kde = False,bins = 50,\n                ax = ax2, color = colors_nude[1], hist_kws ={'alpha':1})\n\n\nax1.set_title(\"Train:Images per Patient Distribution\", fontsize =16)\nax2.set_title(\"Test: Images per Patient Distribution\", fontsize = 16)\nsns.despine(left = True,bottom = True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:08.018744Z","iopub.execute_input":"2022-08-11T08:04:08.019580Z","iopub.status.idle":"2022-08-11T08:04:08.636089Z","shell.execute_reply.started":"2022-08-11T08:04:08.019540Z","shell.execute_reply":"2022-08-11T08:04:08.634776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the files again\ntrain_df.to_csv(\"train_clean.csv\", index = False)\ntest_df.to_csv(\"test_clean.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:08.637671Z","iopub.execute_input":"2022-08-11T08:04:08.638063Z","iopub.status.idle":"2022-08-11T08:04:08.780729Z","shell.execute_reply.started":"2022-08-11T08:04:08.638014Z","shell.execute_reply":"2022-08-11T08:04:08.779046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocess .csv files\n# Add Image Path","metadata":{}},{"cell_type":"code","source":"# Dicom\n# Create the paths\npath_train = directory + '/train/'+ train_df['dcm_name'] + '.dcm'\npath_test = directory + '/test/' + test_df['dcm_name'] + '.dcm'\n\n# append to the original dataframes\ntrain_df['path_dicom'] = path_train\ntest_df['path_dicom'] = path_test\n\n#JPEG\n#create the paths\npath_train = directory + '/jpeg/train/'+ train_df['dcm_name'] + '.jpg'\npath_test = directory + '/jpeg/test/' + test_df['dcm_name'] + '.jpg'\n\n#append to the original dataframe\ntrain_df['path_jpeg'] = path_train\ntest_df['path_jpeg'] = path_test\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:08.782921Z","iopub.execute_input":"2022-08-11T08:04:08.783595Z","iopub.status.idle":"2022-08-11T08:04:08.832141Z","shell.execute_reply.started":"2022-08-11T08:04:08.783552Z","shell.execute_reply":"2022-08-11T08:04:08.831032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_copy = train_df","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:08.834036Z","iopub.execute_input":"2022-08-11T08:04:08.834462Z","iopub.status.idle":"2022-08-11T08:04:08.839788Z","shell.execute_reply.started":"2022-08-11T08:04:08.834425Z","shell.execute_reply":"2022-08-11T08:04:08.838811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# One Hot Encoding","metadata":{}},{"cell_type":"code","source":"# === TRAIN ===\nto_encode = ['sex', 'anatomy', 'diagnosis']\nencoded_all = []\n\nlabel_encoder = LabelEncoder()\n\nfor column in to_encode:\n    encoded = label_encoder.fit_transform(train_df[column])\n    encoded_all.append(encoded)\n    \ntrain_df['sex'] = encoded_all[0]\ntrain_df['anatomy'] = encoded_all[1]\ntrain_df['diagnosis'] = encoded_all[2]\n\nif 'benign_malignant' in train_df.columns : train_df.drop(['benign_malignant'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:08.841444Z","iopub.execute_input":"2022-08-11T08:04:08.842616Z","iopub.status.idle":"2022-08-11T08:04:08.901445Z","shell.execute_reply.started":"2022-08-11T08:04:08.842561Z","shell.execute_reply":"2022-08-11T08:04:08.900168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the files\ntrain_df.to_csv('train_clean.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:08.903338Z","iopub.execute_input":"2022-08-11T08:04:08.904038Z","iopub.status.idle":"2022-08-11T08:04:09.099978Z","shell.execute_reply.started":"2022-08-11T08:04:08.904000Z","shell.execute_reply":"2022-08-11T08:04:09.098964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_copy","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:09.101802Z","iopub.execute_input":"2022-08-11T08:04:09.102527Z","iopub.status.idle":"2022-08-11T08:04:09.124366Z","shell.execute_reply.started":"2022-08-11T08:04:09.102478Z","shell.execute_reply":"2022-08-11T08:04:09.123148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The Images****","metadata":{}},{"cell_type":"code","source":"print('Train .dcm number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/train'))), '\\n' +\n      'Test .dcm number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/test'))), '\\n' +\n      'Train .jpeg number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/jpeg/train'))), '\\n' +\n      'Test .jpeg number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/jpeg/test'))), '\\n' +\n      '-----------------------', '\\n' +\n      'There is the same number of images as in train/ test .csv datasets')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:04:37.577389Z","iopub.execute_input":"2022-08-11T09:04:37.578742Z","iopub.status.idle":"2022-08-11T09:04:41.414551Z","shell.execute_reply.started":"2022-08-11T09:04:37.578692Z","shell.execute_reply":"2022-08-11T09:04:41.413129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Image Shapes**\nAlso, let's look at the size of the images (to not overload the memory, we'll check 100 different images). They are pretty different,","metadata":{}},{"cell_type":"code","source":"shapes_train = []\n\nfor k, path in enumerate(train_df['path_jpeg']):\n    image = Image.open(path)\n    shapes_train.append(image.size)\n    \n    if k >= 100: break\n        \nshapes_train = pd.DataFrame(data = shapes_train, columns = ['H', 'W'], dtype='object')\nshapes_train['Size'] = '[' + shapes_train['H'].astype(str) + ', ' + shapes_train['W'].astype(str) + ']'","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:13:47.643467Z","iopub.execute_input":"2022-08-11T09:13:47.643942Z","iopub.status.idle":"2022-08-11T09:13:47.784616Z","shell.execute_reply.started":"2022-08-11T09:13:47.643895Z","shell.execute_reply":"2022-08-11T09:13:47.783052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (16, 6))\n\na = sns.countplot(shapes_train['Size'], palette=colors_nude)\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n    \nplt.title('100 Images Shapes', fontsize=16)\nsns.despine(left=True, bottom=True);","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:14:00.995401Z","iopub.execute_input":"2022-08-11T09:14:00.995837Z","iopub.status.idle":"2022-08-11T09:14:01.310008Z","shell.execute_reply.started":"2022-08-11T09:14:00.995801Z","shell.execute_reply":"2022-08-11T09:14:01.308655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Dicom Images\n# Malignant vs Benign Images**\nLet's look at the difference between malignant and benign melanomas.","metadata":{}},{"cell_type":"code","source":"def show_images(data, n = 5, rows=1, cols=5, title='Default'):\n    plt.figure(figsize=(16,4))\n\n    for k, path in enumerate(data['path_dicom'][:n]):\n        image = pydicom.read_file(path)\n        image = image.pixel_array\n        \n        # image = resize(image, (200, 200), anti_aliasing=True)\n\n        plt.suptitle(title, fontsize = 16)\n        plt.subplot(rows, cols, k+1)\n        plt.imshow(image)\n        plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:09:11.863333Z","iopub.execute_input":"2022-08-11T09:09:11.863873Z","iopub.status.idle":"2022-08-11T09:09:11.874041Z","shell.execute_reply.started":"2022-08-11T09:09:11.863832Z","shell.execute_reply":"2022-08-11T09:09:11.872527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show Benign Samples\nshow_images(train_df[train_df['target'] == 0], n=10, rows=2, cols=5, title='Benign Sample')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:14:29.838458Z","iopub.execute_input":"2022-08-11T09:14:29.838980Z","iopub.status.idle":"2022-08-11T09:14:51.386118Z","shell.execute_reply.started":"2022-08-11T09:14:29.838937Z","shell.execute_reply":"2022-08-11T09:14:51.384431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show Malignant Samples\nshow_images(train_df[train_df['target'] == 1], n=10, rows=2, cols=5, title='Malignant Sample')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:14:51.388448Z","iopub.execute_input":"2022-08-11T09:14:51.388807Z","iopub.status.idle":"2022-08-11T09:15:07.709890Z","shell.execute_reply.started":"2022-08-11T09:14:51.388773Z","shell.execute_reply":"2022-08-11T09:15:07.708406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Black and White View**","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=6, figsize=(16,6))\nplt.suptitle(\"B&W\", fontsize = 16)\n\nfor i in range(0, 2*6):\n    data = pydicom.read_file(train_df['path_dicom'][i])\n    image = data.pixel_array\n    \n    # Transform to B&W\n    # The function converts an input image from one color space to another.\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n    image = cv2.resize(image, (200,200))\n    \n    x = i // 6\n    y = i % 6\n    axes[x, y].imshow(image, cmap=plt.cm.bone) \n    axes[x, y].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:15:50.271779Z","iopub.execute_input":"2022-08-11T09:15:50.272213Z","iopub.status.idle":"2022-08-11T09:15:55.215760Z","shell.execute_reply.started":"2022-08-11T09:15:50.272177Z","shell.execute_reply":"2022-08-11T09:15:55.214633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Conversion to HSV(Hue,Saturation & Brightness(Value)**","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=6, figsize=(16,6))\nplt.suptitle(\"Conversion to HSV\", fontsize = 16)\n\nfor i in range(0, 2*6):\n    data = pydicom.read_file(train_df['path_dicom'][i])\n    image = data.pixel_array\n    \n    # Transform to B&W\n    # The function converts an input image from one color space to another.\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2HSV)\n    image = cv2.resize(image, (200,200))\n    \n    x = i // 6\n    y = i % 6\n    axes[x, y].imshow(image, cmap=plt.cm.bone) \n    axes[x, y].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:17:35.567699Z","iopub.execute_input":"2022-08-11T09:17:35.568179Z","iopub.status.idle":"2022-08-11T09:17:40.624116Z","shell.execute_reply.started":"2022-08-11T09:17:35.568144Z","shell.execute_reply":"2022-08-11T09:17:40.622508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=6, figsize=(16,6))\nplt.suptitle(\"With Gaussian Blur\", fontsize = 16)\n\nfor i in range(0, 2*6):\n    data = pydicom.read_file(train_df['path_dicom'][i])\n    image = data.pixel_array\n    \n    # Transform to B&W\n    # The function converts an input image from one color space to another.\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2HSV)\n    image = cv2.resize(image, (200,200))\n    image=cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0,0) ,256/10), -4, 128)\n    \n    x = i // 6\n    y = i % 6\n    axes[x, y].imshow(image, cmap=plt.cm.bone) \n    axes[x, y].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:18:49.797040Z","iopub.execute_input":"2022-08-11T09:18:49.797597Z","iopub.status.idle":"2022-08-11T09:18:55.357426Z","shell.execute_reply.started":"2022-08-11T09:18:49.797554Z","shell.execute_reply":"2022-08-11T09:18:55.356101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Conversion to HLS Space(Hue,lightness,saturation)**","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=6, figsize=(16,6))\nplt.suptitle(\"Hue, Saturation, Brightness\", fontsize = 16)\n\nfor i in range(0, 2*6):\n    data = pydicom.read_file(train_df['path_dicom'][i])\n    image = data.pixel_array\n    \n    # Transform to B&W\n    # The function converts an input image from one color space to another.\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2HLS)\n    image = cv2.resize(image, (200,200))\n    \n    x = i // 6\n    y = i % 6\n    axes[x, y].imshow(image, cmap=plt.cm.bone) \n    axes[x, y].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:20:10.877786Z","iopub.execute_input":"2022-08-11T09:20:10.878257Z","iopub.status.idle":"2022-08-11T09:20:16.556677Z","shell.execute_reply.started":"2022-08-11T09:20:10.878220Z","shell.execute_reply":"2022-08-11T09:20:16.555607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **LUV Color Space**","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=6, figsize=(16,6))\nplt.suptitle(\"LUV Color Space\", fontsize = 16)\n\nfor i in range(0, 2*6):\n    data = pydicom.read_file(train_df['path_dicom'][i])\n    image = data.pixel_array\n    \n    # Transform to B&W\n    # The function converts an input image from one color space to another.\n    image = cv2.cvtColor(image, cv2.COLOR_RGB2LUV)\n    image = cv2.resize(image, (200,200))\n    \n    x = i // 6\n    y = i % 6\n    axes[x, y].imshow(image, cmap=plt.cm.bone) \n    axes[x, y].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T09:20:45.506767Z","iopub.execute_input":"2022-08-11T09:20:45.507298Z","iopub.status.idle":"2022-08-11T09:20:51.033602Z","shell.execute_reply.started":"2022-08-11T09:20:45.507257Z","shell.execute_reply":"2022-08-11T09:20:51.032294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Torchvision.transforms**\nIt's a library that goes hand in hand with **PyTorch** and it's easily used to augment data. ","metadata":{}},{"cell_type":"code","source":"#necessary imports \nimport torch\nfrom torch.utils.data import DataLoader,Dataset\nimport torchvision.transforms as transforms\nimport torchvision","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:09.125927Z","iopub.execute_input":"2022-08-11T08:04:09.126670Z","iopub.status.idle":"2022-08-11T08:04:11.379159Z","shell.execute_reply.started":"2022-08-11T08:04:09.126630Z","shell.execute_reply":"2022-08-11T08:04:11.377858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#select a small sample of the jpeg image paths\nimage_list = train_df.sample(12)['path_jpeg']\nprint(image_list)\nimage_list = image_list.reset_index()['path_jpeg']\nprint(image_list)\n\n#show the sample\nplt.figure(figsize = (16,6))\nplt.suptitle(\"Original View\", fontsize = 16)\n\nfor k,image_path in enumerate(image_list):\n    image = mpimg.imread(image_path)\n    \n    plt.subplot(2,6,k+1)\n    plt.imshow(image)\n    plt.axis(\"off\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:11.389525Z","iopub.execute_input":"2022-08-11T08:04:11.390297Z","iopub.status.idle":"2022-08-11T08:04:36.000836Z","shell.execute_reply.started":"2022-08-11T08:04:11.390255Z","shell.execute_reply":"2022-08-11T08:04:35.999354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create Pytorch Dataset object\nclass DatasetExample(Dataset):\n    def __init__(self, image_list, transforms = None):\n        self.image_list = image_list\n        self.transforms = transforms\n        \n    def __len__(self):\n        return (len(self.image_list))\n    \n    # for indexing\n    def __getitem__(self, i):\n        #Read an image\n        image = plt.imread(self.image_list[i])\n        image = Image.fromarray(image).convert(\"RGB\")\n        image = np.asarray(image).astype(np.uint8)\n        if self.transforms is not None:\n            image = self.transforms(image)\n        \n        return torch.tensor(image,dtype = torch.float)\n    \n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:36.002424Z","iopub.execute_input":"2022-08-11T08:04:36.002867Z","iopub.status.idle":"2022-08-11T08:04:36.013583Z","shell.execute_reply.started":"2022-08-11T08:04:36.002827Z","shell.execute_reply":"2022-08-11T08:04:36.011829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predefined show images function\n\ndef show_transform(image, title = \"Default\"):\n    plt.figure(figsize = (16,6))\n    plt.suptitle(title, fontsize = 16)\n    \n    #Unnormalize\n    image = image/ 2 + 0.5\n    npimg = image.numpy()\n    npimg = np.clip(npimg,0, 1)\n    plt.imshow(np.transpose(npimg,(1,2,0)))\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:36.016030Z","iopub.execute_input":"2022-08-11T08:04:36.016485Z","iopub.status.idle":"2022-08-11T08:04:36.029807Z","shell.execute_reply.started":"2022-08-11T08:04:36.016446Z","shell.execute_reply":"2022-08-11T08:04:36.028429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Transform\ntransform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((300, 300)),\n    transforms.CenterCrop((100, 100)),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5,0.5,0.5),(0.5,0.5,0.5))\n])\n\n# Create the dataset\npytorch_dataset = DatasetExample(image_list= image_list, transforms = transform)\npytorch_dataloader = DataLoader(dataset = pytorch_dataset, batch_size =12,shuffle = True)\n\n#select the data\nimages = next(iter(pytorch_dataloader))\n\n#show images\nshow_transform(torchvision.utils.make_grid(images, nrow = 6), title = \"Crop\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:36.031746Z","iopub.execute_input":"2022-08-11T08:04:36.033102Z","iopub.status.idle":"2022-08-11T08:04:46.686831Z","shell.execute_reply.started":"2022-08-11T08:04:36.033058Z","shell.execute_reply":"2022-08-11T08:04:46.685494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **ColorJitter**\nRandomly change the brightness, contrast and saturation of an image","metadata":{}},{"cell_type":"code","source":"#Transform\ntransform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((300, 300)),\n    transforms.ColorJitter(brightness =0.7,contrast = 0.7,saturation =0.7,hue = 0.5),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5,0.5,0.5),(0.5,0.5,0.5))\n])\n\n# Create the dataset\npytorch_dataset = DatasetExample(image_list= image_list, transforms = transform)\npytorch_dataloader = DataLoader(dataset = pytorch_dataset, batch_size =12,shuffle = True)\n\n#select the data\nimages = next(iter(pytorch_dataloader))\n\n#show images\nshow_transform(torchvision.utils.make_grid(images, nrow = 6), title = \"Color Jitter\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:46.688859Z","iopub.execute_input":"2022-08-11T08:04:46.689369Z","iopub.status.idle":"2022-08-11T08:04:57.602561Z","shell.execute_reply.started":"2022-08-11T08:04:46.689297Z","shell.execute_reply":"2022-08-11T08:04:57.601471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **RandomGreyScale**\nRandomly convert image to greyscale with a probability of p","metadata":{}},{"cell_type":"code","source":"#Transform\ntransform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((300, 300)),\n    transforms.RandomGrayscale(p =0.7),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5,0.5,0.5),(0.5,0.5,0.5))\n])\n\n# Create the dataset\npytorch_dataset = DatasetExample(image_list= image_list, transforms = transform)\npytorch_dataloader = DataLoader(dataset = pytorch_dataset, batch_size =12,shuffle = True)\n\n#select the data\nimages = next(iter(pytorch_dataloader))\n\n#show images\nshow_transform(torchvision.utils.make_grid(images, nrow = 6), title = \"Random Greyscale\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:04:57.604557Z","iopub.execute_input":"2022-08-11T08:04:57.605266Z","iopub.status.idle":"2022-08-11T08:05:07.210267Z","shell.execute_reply.started":"2022-08-11T08:04:57.605224Z","shell.execute_reply":"2022-08-11T08:05:07.208655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **RandomVerticalFlip**\nVertically flip the given PIL Image randomly with a given probability.","metadata":{}},{"cell_type":"code","source":"#Transform\ntransform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((300, 300)),\n    transforms.RandomVerticalFlip(p =0.7),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5,0.5,0.5),(0.5,0.5,0.5))\n])\n\n# Create the dataset\npytorch_dataset = DatasetExample(image_list= image_list, transforms = transform)\npytorch_dataloader = DataLoader(dataset = pytorch_dataset, batch_size =12,shuffle = True)\n\n#select the data\nimages = next(iter(pytorch_dataloader))\n\n#show images\nshow_transform(torchvision.utils.make_grid(images, nrow = 6), title = \"Random Vertical Flip\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:05:07.212345Z","iopub.execute_input":"2022-08-11T08:05:07.212748Z","iopub.status.idle":"2022-08-11T08:05:17.466227Z","shell.execute_reply.started":"2022-08-11T08:05:07.212711Z","shell.execute_reply":"2022-08-11T08:05:17.464840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Hair Removal**\nAs you may have noticed, all images are for light skin colors, so no preprocessing in this area is required. However, hair removal might be a good augmentation that will help the model perform better.","metadata":{}},{"cell_type":"code","source":"BASE_PATH = \"../input/siim-isic-melanoma-classification\"\nhair_images =['ISIC_0078712','ISIC_0080817','ISIC_0082348','ISIC_0109869','ISIC_0155012','ISIC_0159568','ISIC_0164145','ISIC_0194550','ISIC_0194914','ISIC_0202023']\nwithout_hair_images = ['ISIC_0015719','ISIC_0074268','ISIC_0075914','ISIC_0084395','ISIC_0085718','ISIC_0081956']\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:05:17.468696Z","iopub.execute_input":"2022-08-11T08:05:17.469191Z","iopub.status.idle":"2022-08-11T08:05:17.475820Z","shell.execute_reply.started":"2022-08-11T08:05:17.469149Z","shell.execute_reply":"2022-08-11T08:05:17.474780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = len(hair_images[:8])\n\n\nfig =plt.figure(figsize = (20,30))\n\n\nfor i,image_name in enumerate(hair_images[:8]):\n    image =cv2.imread(BASE_PATH + '/jpeg/train/' + image_name +'.jpg')\n    image_resize = cv2.resize(image, (1024,1024))\n    plt.subplot(l, 5, (i*5)+1)\n    \n    #Convert original image to RGB\n    plt.imshow(cv2.cvtColor(image_resize, cv2.COLOR_BGR2RGB))\n    plt.axis(\"off\")\n    plt.title(\"Original: \"+ image_name)\n    \n    #convert image to grayscale\n    grayScale = cv2.cvtColor(image_resize,cv2.COLOR_RGB2GRAY)\n    plt.subplot(l,5,(i*5)+2)\n    plt.imshow(grayScale)\n    plt.axis(\"off\")\n    plt.title(\"GrayScale: \" + image_name)\n    \n    #kernel for morphological filtering\n    kernel = cv2.getStructuringElement(1,(17,17))\n    \n    #perform blackhat filtering on the grayscale image to find hair contours\n    blackhat = cv2.morphologyEx(grayScale,cv2.MORPH_BLACKHAT,kernel)\n    plt.subplot(l,5, (i*5)+ 3)\n    plt.imshow(blackhat)\n    plt.axis(\"off\")\n    plt.title(\"blackhat: \" + image_name)\n    \n    \n    #intensify the hair contours in preparation for the inpainting\n    ret, threshold = cv2.threshold(blackhat, 10,255, cv2.THRESH_BINARY)\n    plt.subplot(l,5,(i*5) + 4)\n    plt.imshow(threshold)\n    plt.axis(\"off\")\n    plt.title(\"Threshold: \" + image_name)\n    \n    #inpaint the original image depending on the mask\n    final_image = cv2.inpaint(image_resize,threshold, 1, cv2.INPAINT_TELEA)\n    plt.subplot(l,5, (i*5) +5)\n    plt.imshow(cv2.cvtColor(final_image, cv2.COLOR_BGR2RGB))\n    plt.axis(\"off\")\n    plt.title(\"Final Image: \" + image_name)\n    \n\nplt.plot()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:05:17.477441Z","iopub.execute_input":"2022-08-11T08:05:17.477818Z","iopub.status.idle":"2022-08-11T08:07:05.644603Z","shell.execute_reply.started":"2022-08-11T08:05:17.477784Z","shell.execute_reply":"2022-08-11T08:07:05.643391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def hair_remove(image):\n    #convert image to greyscale\n    grayScale = cv2.cvtColor(image,cv2.COLOR_RGB2GRAY)\n    \n    #kernel for morphologyEx\n    kernel = cv2.getStructuringElement(1,(17,17))\n    \n    #apply MORPH_Blackhat to grayscale image\n    blackhat = cv2.morphologyEx(grayScale,cv2.MORPH_BLACKHAT,kernel)\n    #apply thresholding to blackhat\n    _, threshold = cv2.threshold(blackhat,10,255,cv2.THRESH_BINARY)\n    #inpaint with original image and threshold image\n    final_image = cv2.inpaint(image,threshold,1,cv2.INPAINT_TELEA)\n    \n    return final_image\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:05.645852Z","iopub.execute_input":"2022-08-11T08:07:05.646495Z","iopub.status.idle":"2022-08-11T08:07:05.654698Z","shell.execute_reply.started":"2022-08-11T08:07:05.646454Z","shell.execute_reply":"2022-08-11T08:07:05.653485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Select a small sample of the .jpeg image paths\n# we select some hairy photos on purpose\nhairy_photos = train_df[train_df[\"sex\"]==1].reset_index().iloc[[12, 14,17, 22, 33, 34]]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:05.656536Z","iopub.execute_input":"2022-08-11T08:07:05.656867Z","iopub.status.idle":"2022-08-11T08:07:05.680949Z","shell.execute_reply.started":"2022-08-11T08:07:05.656835Z","shell.execute_reply":"2022-08-11T08:07:05.679830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hairy_photos","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:05.682544Z","iopub.execute_input":"2022-08-11T08:07:05.683428Z","iopub.status.idle":"2022-08-11T08:07:05.701821Z","shell.execute_reply.started":"2022-08-11T08:07:05.683387Z","shell.execute_reply":"2022-08-11T08:07:05.700354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_list = hairy_photos['path_jpeg']\nimage_list = image_list.reset_index()['path_jpeg']\nimage_list","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:05.703676Z","iopub.execute_input":"2022-08-11T08:07:05.704472Z","iopub.status.idle":"2022-08-11T08:07:05.715601Z","shell.execute_reply.started":"2022-08-11T08:07:05.704430Z","shell.execute_reply":"2022-08-11T08:07:05.714613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show the augmented images\nplt.figure(figsize =(16,3))\nplt.suptitle(\"Original Hairy Images\" , fontsize = 16)\n\nfor k,path in enumerate(image_list):\n    image = mpimg.imread(path)\n    image = cv2.resize(image, (300,300))\n    \n    plt.subplot(1,6, k+1)\n    plt.imshow(image)\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:05.716963Z","iopub.execute_input":"2022-08-11T08:07:05.717921Z","iopub.status.idle":"2022-08-11T08:07:08.903777Z","shell.execute_reply.started":"2022-08-11T08:07:05.717821Z","shell.execute_reply":"2022-08-11T08:07:08.902592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show the sample\nplt.figure(figsize =(16,3))\nplt.suptitle(\"Non Hairy Images\" , fontsize = 16)\n\nfor k,path in enumerate(image_list):\n    image = mpimg.imread(path)\n    image = cv2.resize(image, (300,300))\n    image = hair_remove(image)\n    \n    plt.subplot(1,6, k+1)\n    plt.imshow(image)\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:08.905376Z","iopub.execute_input":"2022-08-11T08:07:08.905734Z","iopub.status.idle":"2022-08-11T08:07:13.609001Z","shell.execute_reply.started":"2022-08-11T08:07:08.905702Z","shell.execute_reply":"2022-08-11T08:07:13.607667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Diagnosis Visualization**\n**We have seven types of diagnosed cancerous growths here:**\n\n* Unkown: a possibly novel type of growth\n* Nevus: (from Google) a usually non-cancerous disorder of pigment-producing skin cells commonly called birth marks or moles.\n* Melanoma: Skin cancer's form (what we are working with)\n* Seborrheic keratosis: Brown, waxy and patchy growths that are not related to skin cancer.\n* Lentigo NOS: A type of skin cancer that starts from the outside of the skin and attacks by going inword.\n* Lichenoid keratosis: It is a thin pigmented sort of plaque, if you will.\n* Solar lentigo: Like lentigo but caused by UV rays from the sun (very common in Delhi)\n* cafe-au-lait macule: French for \"coffee with milk\". These are brownish spots also called \"giraffe spots\".\n* atypical melanocytic proliferation: Abnormal quantities of melanin appear on the skin.","metadata":{}},{"cell_type":"code","source":"train_images_dir = \"../input/siim-isic-melanoma-classification/train\"\ndef view_images(images, title =\"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height, width, figsize =(15,5))\n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir,list(images)[im] + '.dcm'))\n        image = data.pixel_array\n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image,cmap = plt.cm.bone)\n        axs[i,j].axis('off')\n    \n    plt.subplots_adjust(wspace =0,hspace =0)\n    plt.suptitle(title)\n    plt.show()    ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:13.610617Z","iopub.execute_input":"2022-08-11T08:07:13.611991Z","iopub.status.idle":"2022-08-11T08:07:13.622828Z","shell.execute_reply.started":"2022-08-11T08:07:13.611941Z","shell.execute_reply":"2022-08-11T08:07:13.621504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_1 = pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\")\ntrain_df_1","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:13.624173Z","iopub.execute_input":"2022-08-11T08:07:13.624530Z","iopub.status.idle":"2022-08-11T08:07:13.702350Z","shell.execute_reply.started":"2022-08-11T08:07:13.624499Z","shell.execute_reply":"2022-08-11T08:07:13.701028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_images(train_df_1[train_df_1['diagnosis']=='nevus']['image_name'], title = \"Nevus Pigmentous Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:13.704435Z","iopub.execute_input":"2022-08-11T08:07:13.704936Z","iopub.status.idle":"2022-08-11T08:07:25.673802Z","shell.execute_reply.started":"2022-08-11T08:07:13.704889Z","shell.execute_reply":"2022-08-11T08:07:25.672386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From Nevus pigmentatious growth, we can observe that nevus also has an orange halo which is not so visceral and pronounced. This can be confusing as our model can mix up this orange halo and the other orange halo","metadata":{}},{"cell_type":"code","source":"view_images(train_df_1[train_df_1['diagnosis']=='melanoma']['image_name'], title = \"Melanoma's Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:07:25.675719Z","iopub.execute_input":"2022-08-11T08:07:25.676080Z","iopub.status.idle":"2022-08-11T08:08:03.357892Z","shell.execute_reply.started":"2022-08-11T08:07:25.676047Z","shell.execute_reply":"2022-08-11T08:08:03.356364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that is cancer. Melanoma in most cases displays the vicious and visceral orange halo as it devours alive other poor cells who where unlucky enough to cross its path.","metadata":{}},{"cell_type":"code","source":"train_df_1.diagnosis.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:08:03.359586Z","iopub.execute_input":"2022-08-11T08:08:03.359964Z","iopub.status.idle":"2022-08-11T08:08:03.373272Z","shell.execute_reply.started":"2022-08-11T08:08:03.359931Z","shell.execute_reply":"2022-08-11T08:08:03.371910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_images(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title = \"Lentigo's Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:08:03.375209Z","iopub.execute_input":"2022-08-11T08:08:03.375668Z","iopub.status.idle":"2022-08-11T08:08:38.930644Z","shell.execute_reply.started":"2022-08-11T08:08:03.375631Z","shell.execute_reply":"2022-08-11T08:08:38.928869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lentigo has not only the orange halo of death but also the green halo, which could be what distinguishes lentigo from melanoma. This green halo could be a special characteristic of lentigo that could help our model to distinguish lentigo from other carcinogenic growths.","metadata":{}},{"cell_type":"code","source":"view_images(train_df_1[train_df_1['diagnosis']=='lichenoid keratosis']['image_name'], title = \"Lichenoid's Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:08:38.932764Z","iopub.execute_input":"2022-08-11T08:08:38.933203Z","iopub.status.idle":"2022-08-11T08:09:12.854124Z","shell.execute_reply.started":"2022-08-11T08:08:38.933164Z","shell.execute_reply":"2022-08-11T08:09:12.853077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lichenoid mimics the dangers of real carcinogenous growth with its halo(s). Lichenoid is very good at mimicking other forms of cancer.","metadata":{}},{"cell_type":"code","source":"view_images(train_df_1[train_df_1['diagnosis']=='seborrheic keratosis']['image_name'], title = \"Seborrheic's Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:09:12.855400Z","iopub.execute_input":"2022-08-11T08:09:12.856575Z","iopub.status.idle":"2022-08-11T08:09:41.032153Z","shell.execute_reply.started":"2022-08-11T08:09:12.856521Z","shell.execute_reply":"2022-08-11T08:09:41.030942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seborrheic and lichenoid keratosis are both rather similar.","metadata":{}},{"cell_type":"code","source":"def view_one_image(images, title =\"\", aug = None):\n    width = 1\n    height = 1\n    fig,axs = plt.subplots(height, width, figsize =(15,5))\n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir,list(images)[im] + '.dcm'))\n        image = data.pixel_array\n        axs.imshow(image,cmap = plt.cm.bone)\n        axs.axis('off')\n    \n    plt.subplots_adjust(wspace =0,hspace =0)\n    plt.suptitle(title)\n    plt.show() \n    \nview_one_image(train_df_1[train_df_1['diagnosis']=='atypical melanocytic proliferation']['image_name'], title = \"Atypical Melanocytic's Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:09:41.033895Z","iopub.execute_input":"2022-08-11T08:09:41.034899Z","iopub.status.idle":"2022-08-11T08:09:44.319422Z","shell.execute_reply.started":"2022-08-11T08:09:41.034853Z","shell.execute_reply":"2022-08-11T08:09:44.318035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert to **LAB Colorspace** and **CLAHE** applied on a single channel","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title =\"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height, width, figsize =(15,15))\n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir,list(images)[im] + '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2LAB)\n        image = cv2.resize(image, (256,256))\n        clahe = cv2.createCLAHE(clipLimit =2.0 , tileGridSize = (256,256))\n        image[:,:,0] = clahe.apply(image[:,:,0])\n        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image,cmap = plt.cm.bone)\n        axs[i,j].axis('off')\n    \n    plt.subplots_adjust(wspace =0,hspace =0)\n    plt.suptitle(title)\n\n\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title = \"Lentigo's Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:09:44.320740Z","iopub.execute_input":"2022-08-11T08:09:44.321057Z","iopub.status.idle":"2022-08-11T08:09:59.327913Z","shell.execute_reply.started":"2022-08-11T08:09:44.321028Z","shell.execute_reply":"2022-08-11T08:09:59.326069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Grayscale Images**\nWe will first try to visualize in grayscale (only gray colors) so that it is possible for us to clearly visualize the varied differences in color, region, and shape.","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title =\"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height, width, figsize =(15,15))\n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir,list(images)[im] + '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n        image = cv2.resize(image, (256,256))        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image,cmap = plt.cm.bone)\n        axs[i,j].axis('off')\n    \n    plt.subplots_adjust(wspace =0,hspace =0)\n    plt.suptitle(title)\n\n\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title = \"Lentigo's Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:09:59.329628Z","iopub.execute_input":"2022-08-11T08:09:59.329942Z","iopub.status.idle":"2022-08-11T08:10:12.456076Z","shell.execute_reply.started":"2022-08-11T08:09:59.329913Z","shell.execute_reply":"2022-08-11T08:10:12.454818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Neuron Engineer's Method**","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title =\"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height, width, figsize =(15,15))\n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir,list(images)[im] + '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) #conversion from BGR image to RGB channel image\n        image = cv2.resize(image, (256,256)) \n        image= cv2.addWeighted ( image,4, cv2.GaussianBlur( image , (0,0) , 10) ,-4 ,128) #adding Gaussian Blur to the images\n\n        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image,cmap = plt.cm.bone)\n        axs[i,j].axis('off')\n    \n    plt.subplots_adjust(wspace =0,hspace =0)\n    plt.suptitle(title)\n    \nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title = \"Lentigo's Growth\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:10:12.458210Z","iopub.execute_input":"2022-08-11T08:10:12.458915Z","iopub.status.idle":"2022-08-11T08:10:23.210128Z","shell.execute_reply.started":"2022-08-11T08:10:12.458873Z","shell.execute_reply":"2022-08-11T08:10:23.208836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Circle Crop**","metadata":{}},{"cell_type":"code","source":"def crop_image_from_gray(img, tol =7):\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img > tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1), mask.any(0))].shape[0]\n        if (check_shape == 0): #image is too dark that we crop out everything\n            return img # return original image\n        else:\n            img1 = img[:,:,0][np.ix_(mask.any(1), mask.any(0))]\n            img2 = img[:,:,1][np.ix_(mask.any(1), mask.any(0))]\n            img3 = img[:,:,2][np.ix_(mask.any(1), mask.any(0))]\n            img = np.stack([img1,img2,img3], axis = -1)\n    return img\n    \ndef circle_crop(img, sigmaX = 10):\n    \"Create circular crop around image centre\"\n    img = crop_image_from_gray(img)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height,width,depth = img.shape\n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height,width), np.uint8)\n    cv2.circle(circle_img,(x,y), int(r), 1, thickness = -1)\n    img = cv2.bitwise_and(img,img, mask = circle_img)\n    img = crop_image_from_gray(img)\n    \n    img = cv2.addWeighted(img, 4,cv2.GaussianBlur(img,(0,0),sigmaX) , -4, 128)\n    \n    return img\n\ndef view_images_aug(images, title =\"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height, width, figsize =(15,15))\n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir,list(images)[im] + '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) #conversion from BGR image to RGB channel image\n        image = cv2.resize(image, (256,256)) \n        image = circle_crop(image)\n\n        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image,cmap = plt.cm.bone)\n        axs[i,j].axis('off')\n    \n    plt.subplots_adjust(wspace =0,hspace =0)\n    plt.suptitle(title)\n    \nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title = \"Lentigo's Growth\")   ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:10:23.211854Z","iopub.execute_input":"2022-08-11T08:10:23.212236Z","iopub.status.idle":"2022-08-11T08:10:33.590403Z","shell.execute_reply.started":"2022-08-11T08:10:23.212199Z","shell.execute_reply":"2022-08-11T08:10:33.588766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Circular crop has successfully worked, although it may not be feasible for images where the tumor is on the edge of the image.","metadata":{}},{"cell_type":"markdown","source":"# **Auto Cropping**","metadata":{}},{"cell_type":"code","source":"def crop_image_from_gray(img, tol =7):\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img > tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1), mask.any(0))].shape[0]\n        if (check_shape == 0): #image is too dark that we crop out everything\n            return img # return original image\n        else:\n            img1 = img[:,:,0][np.ix_(mask.any(1), mask.any(0))]\n            img2 = img[:,:,1][np.ix_(mask.any(1), mask.any(0))]\n            img3 = img[:,:,2][np.ix_(mask.any(1), mask.any(0))]\n            img = np.stack([img1,img2,img3], axis = -1)\n    return img\n\ndef view_images_aug(images, title =\"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height, width, figsize =(15,15), constrained_layout=True)\n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir,list(images)[im] + '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) #conversion from BGR image to RGB channel image\n        image = cv2.resize(image, (256,256)) \n        image = crop_image_from_gray(image)\n\n        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image,cmap = plt.cm.bone)\n        axs[i,j].axis('off')\n    \n    plt.subplots_adjust(wspace =0,hspace =0)\n    plt.suptitle(title)\n    \nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title = \"Lentigo's Growth\")   ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:10:33.592952Z","iopub.execute_input":"2022-08-11T08:10:33.593446Z","iopub.status.idle":"2022-08-11T08:10:44.114379Z","shell.execute_reply.started":"2022-08-11T08:10:33.593403Z","shell.execute_reply":"2022-08-11T08:10:44.112986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Background Subtraction**\nAnother thing you can do is Background Subtraction.","metadata":{}},{"cell_type":"code","source":"fgbg = cv2.createBackgroundSubtractorMOG2()\ndef view_images_aug(images, title = '', aug = None):\n    width = 6\n    height = 5\n    fig, axs = plt.subplots(height, width, figsize=(15,15), constrained_layout=True)\n    for im in range(0, height * width):  \n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm')) \n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image= fgbg.apply(image)\n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n        \n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:10:44.116020Z","iopub.execute_input":"2022-08-11T08:10:44.116500Z","iopub.status.idle":"2022-08-11T08:10:57.797115Z","shell.execute_reply.started":"2022-08-11T08:10:44.116461Z","shell.execute_reply":"2022-08-11T08:10:57.795713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Image Segmentation**","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title = \"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height,width,figsize =(15,15))\n    \n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm')) \n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n        ret, threshold = cv2.threshold(image,0,255,cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n    plt.suptitle(title)\n\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:10:57.798667Z","iopub.execute_input":"2022-08-11T08:10:57.798974Z","iopub.status.idle":"2022-08-11T08:11:17.225576Z","shell.execute_reply.started":"2022-08-11T08:10:57.798946Z","shell.execute_reply":"2022-08-11T08:11:17.224129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is called **image segmentation.** It breals down an image into its constituent parts represented by the distinction between regions (i.e **in this instance the growth is white and the halo/surrrounding area is blackened)**. It helps our model to visually understand the distinctions even better than using grayscale images and it also helps to let our model identify the tumor.\n\nHowever, it has a heavy downside as we lose all information inside and outside the growth and thus our model loses the capability to understand or learn something from the image.","metadata":{}},{"cell_type":"markdown","source":"# **A finer method for image segmentation**","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title = \"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height,width,figsize =(15,15))\n    \n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm')) \n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n        ret, threshold = cv2.threshold(image,0,255,cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n        kernel = np.ones((3,3), np.uint8)\n        opening = cv2.morphologyEx(threshold,cv2.MORPH_OPEN,kernel, iterations = 2)\n        \n        #sure background area\n        sure_bg = cv2.dilate(opening, kernel, iterations = 3)\n        \n        #Finding sure foreground area\n        dist_transform = cv2.distanceTransform(opening,cv2.DIST_L2, 5)\n        ret, sure_fg = cv2.threshold(dist_transform, 0.7*dist_transform.max(), 255, 0)\n        \n        #finding unknown area\n        sure_fg = np.uint8(sure_fg)\n        unknown = cv2.subtract( sure_bg, sure_fg)\n        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:11:17.227249Z","iopub.execute_input":"2022-08-11T08:11:17.227610Z","iopub.status.idle":"2022-08-11T08:11:38.301853Z","shell.execute_reply.started":"2022-08-11T08:11:17.227580Z","shell.execute_reply":"2022-08-11T08:11:38.300368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Grayscale Image Segmentation**\n# Watershed Algorithm**","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title = \"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height,width,figsize =(15,15))\n    \n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm')) \n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n        ret, threshold = cv2.threshold(image,0,255,cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n        kernel = np.ones((3,3), np.uint8)\n        opening = cv2.morphologyEx(threshold,cv2.MORPH_OPEN,kernel, iterations = 2)\n        \n        #sure background area\n        sure_bg = cv2.dilate(opening, kernel, iterations = 3)\n        \n        #Finding sure foreground area\n        dist_transform = cv2.distanceTransform(opening,cv2.DIST_L2, 5)\n        ret, sure_fg = cv2.threshold(dist_transform, 0.7*dist_transform.max(), 255, 0)\n        \n        #finding unknown area\n        sure_fg = np.uint8(sure_fg)\n        unknown = cv2.subtract( sure_bg, sure_fg)\n        ret, markers = cv2.connectedComponents(sure_fg)\n        \n        # Add one to all labels so that sure background is not 0, but 1\n        markers = markers + 1\n        \n        markers[unknown == 255] =0\n        image = cv2.cvtColor(image, cv2.COLOR_GRAY2BGR)\n        \n        markers = cv2.watershed(image, markers)\n        image[markers == -1] = [255,0,0]\n        \n        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n    plt.suptitle(title)\n\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:11:38.303650Z","iopub.execute_input":"2022-08-11T08:11:38.304165Z","iopub.status.idle":"2022-08-11T08:12:24.513266Z","shell.execute_reply.started":"2022-08-11T08:11:38.304130Z","shell.execute_reply":"2022-08-11T08:12:24.511685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These are now the segmented grayscale images, complete with markers. It will be a bit difficult for the model to learn anything from this due to the complete and utter confusion (pardon me) in the image with regard to the clear distinctions between image segments and there is no disctinction between parts of the image","metadata":{}},{"cell_type":"markdown","source":"# **Fourier Method for Pixel Distribution**","metadata":{}},{"cell_type":"code","source":"from numpy.fft import *\n\ndef view_images_aug(images, title = \"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height,width,figsize =(15,15))\n    \n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm')) \n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n        f = np.fft.fft2(image)\n        fshift = np.fft.fftshift(f)\n        magnitude_spectrum = 20*np.log(np.abs(fshift))\n        \n        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(magnitude_spectrum, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n    plt.suptitle(title)\n\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:12:24.514964Z","iopub.execute_input":"2022-08-11T08:12:24.515333Z","iopub.status.idle":"2022-08-11T08:13:27.116946Z","shell.execute_reply.started":"2022-08-11T08:12:24.515285Z","shell.execute_reply":"2022-08-11T08:13:27.115817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is helpful to understand where the majority of the growth is concentrated.","metadata":{}},{"cell_type":"markdown","source":"# **Albumentations library demonstration**","metadata":{}},{"cell_type":"code","source":"import albumentations as A\nimage_folder_path = \"../input/siim-isic-melanoma-classification/jpeg/train\"\nchosen_image = cv2.imread(os.path.join(image_folder_path, \"ISIC_0079038.jpg\"))\nalbumentation_list = [A.RandomSunFlare(p =1),A.RandomFog(p=1),A.RandomBrightness(p=1),\n                     A.RandomCrop(p=1,height=512,width=512),A.Rotate(p=1,limit =90),\n                     A.RGBShift(p=1),A.RandomSnow(p =1),\n                     A.HorizontalFlip(p=1),A.VerticalFlip(p=1),A.RandomContrast(limit =0.5, p=1),\n                     A.HueSaturationValue(p=1,hue_shift_limit=20, sat_shift_limit=30, val_shift_limit=50)]\n\nimg_matrix_list =[]\nbboxes_list =[]\n\nfor aug_type in albumentation_list:\n    img = aug_type(image = chosen_image)['image']\n    img_matrix_list.append(img)\n\nimg_matrix_list.insert(0, chosen_image)\n\ntitles_list = [\"Original\",\"RandomSunFlare\",\"RandomFog\",\"RandomBrightness\",\n               \"RandomCrop\",\"Rotate\", \"RGBShift\", \"RandomSnow\",\"HorizontalFlip\", \n               \"VerticalFlip\", \"RandomContrast\",\"HSV\"]\n\n\n#helper function\ndef plot_multiple_img(img_matrix_list,title_list, ncols, main_title =\"\"):\n    fig,myaxes = plt.subplots(figsize =(20,15),nrows = 3, ncols = ncols ,squeeze = False)\n    fig.suptitle(main_title,fontsize = 30)\n    fig.subplots_adjust(wspace = 0.3)\n    fig.subplots_adjust(hspace = 0.3)\n    \n    for i, (img, title) in enumerate(zip(img_matrix_list,title_list)):\n        myaxes[i // ncols][i % ncols].imshow(img)\n        myaxes[i // ncols][i % ncols].set_title(title, fontsize = 15)\n        \n    plt.show()\n\nplot_multiple_img(img_matrix_list,titles_list, ncols =4, main_title =\"Different Types of Augmentations\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:13:27.118451Z","iopub.execute_input":"2022-08-11T08:13:27.119898Z","iopub.status.idle":"2022-08-11T08:14:33.316938Z","shell.execute_reply.started":"2022-08-11T08:13:27.119817Z","shell.execute_reply":"2022-08-11T08:14:33.315472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have much more augmentations we can try like:","metadata":{}},{"cell_type":"code","source":"image_folder_path = \"../input/siim-isic-melanoma-classification/jpeg/train\"\nchosen_image = cv2.imread(os.path.join(image_folder_path, \"ISIC_0079038.jpg\"))\nalbumentation_list = [A.RandomSunFlare(p =1),A.GaussNoise(p=1),A.CLAHE(p=1),\n                     A.RandomRain(p=1),A.Rotate(p=1,limit =90),\n                     A.RGBShift(p=1),A.RandomSnow(p =1),\n                     A.HorizontalFlip(p=1),A.VerticalFlip(p=1),A.RandomContrast(limit =0.5, p=1),\n                     A.HueSaturationValue(p=1,hue_shift_limit=20, sat_shift_limit=30, val_shift_limit=50)]\n\nimg_matrix_list =[]\nbboxes_list =[]\n\nfor aug_type in albumentation_list:\n    img = aug_type(image = chosen_image)['image']\n    img_matrix_list.append(img)\n\nimg_matrix_list.insert(0, chosen_image)\n\ntitles_list = [\"Original\",\"RandomSunFlare\",\"GaussNoise\",\"CLAHE\",\n               \"RandomRain\",\"Rotate\", \"RGBShift\", \"RandomSnow\",\n               \"HorizontalFlip\", \"VerticalFlip\", \"RandomContrast\",\"HSV\"]\n\n\n#helper function\ndef plot_multiple_img(img_matrix_list,title_list, ncols, main_title =\"\"):\n    fig,myaxes = plt.subplots(figsize =(20,15),nrows = 3, ncols = ncols ,squeeze = False)\n    fig.suptitle(main_title,fontsize = 30)\n    fig.subplots_adjust(wspace = 0.3)\n    fig.subplots_adjust(hspace = 0.3)\n    \n    for i, (img, title) in enumerate(zip(img_matrix_list,title_list)):\n        myaxes[i // ncols][i % ncols].imshow(img)\n        myaxes[i // ncols][i % ncols].set_title(title, fontsize = 15)\n        \n    plt.show()\n\nplot_multiple_img(img_matrix_list,titles_list, ncols =4, main_title =\"Different Types of Augmentations\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:14:33.318909Z","iopub.execute_input":"2022-08-11T08:14:33.319333Z","iopub.status.idle":"2022-08-11T08:15:09.095984Z","shell.execute_reply.started":"2022-08-11T08:14:33.319278Z","shell.execute_reply":"2022-08-11T08:15:09.094562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Mess Around With p**","metadata":{}},{"cell_type":"code","source":"image_folder_path = \"../input/siim-isic-melanoma-classification/jpeg/train\"\nchosen_image = cv2.imread(os.path.join(image_folder_path, \"ISIC_0079038.jpg\"))\nalbumentation_list = [A.RandomSunFlare(p =0.8),A.GaussNoise(p= 0.8),A.CLAHE(p= 0.9),\n                     A.RandomRain(p=1),A.Rotate(p=1,limit =90),\n                     A.RGBShift(p=1),A.RandomSnow(p =1),\n                     A.HorizontalFlip(p= 0.8),A.VerticalFlip(p= 0.8),A.RandomContrast(limit =0.5, p=1),\n                     A.HueSaturationValue(p=1,hue_shift_limit=20, sat_shift_limit=30, val_shift_limit=50)]\n\nimg_matrix_list =[]\nbboxes_list =[]\n\nfor aug_type in albumentation_list:\n    img = aug_type(image = chosen_image)['image']\n    img_matrix_list.append(img)\n\nimg_matrix_list.insert(0, chosen_image)\n\ntitles_list = [\"Original\",\"RandomSunFlare\",\"GaussNoise\",\"CLAHE\",\n               \"RandomRain\",\"Rotate\", \"RGBShift\", \"RandomSnow\",\n               \"HorizontalFlip\", \"VerticalFlip\", \"RandomContrast\",\"HSV\"]\n\n\n#helper function\ndef plot_multiple_img(img_matrix_list,title_list, ncols, main_title =\"\"):\n    fig,myaxes = plt.subplots(figsize =(20,15),nrows = 3, ncols = ncols ,squeeze = False)\n    fig.suptitle(main_title,fontsize = 30)\n    fig.subplots_adjust(wspace = 0.3)\n    fig.subplots_adjust(hspace = 0.3)\n    \n    for i, (img, title) in enumerate(zip(img_matrix_list,title_list)):\n        myaxes[i // ncols][i % ncols].imshow(img)\n        myaxes[i // ncols][i % ncols].set_title(title, fontsize = 15)\n        \n    plt.show()\n\nplot_multiple_img(img_matrix_list,titles_list, ncols =4, main_title =\"Different Types of Augmentations\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:15:09.097941Z","iopub.execute_input":"2022-08-11T08:15:09.098418Z","iopub.status.idle":"2022-08-11T08:15:41.512457Z","shell.execute_reply.started":"2022-08-11T08:15:09.098369Z","shell.execute_reply":"2022-08-11T08:15:41.511098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Erosion**","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title = \"\", aug = None):\n    width = 6\n    height = 4\n    fig,axs = plt.subplots(height,width,figsize =(15,15),constrained_layout=True)\n    \n    for im in range(0, height*width):\n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm')) \n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        \n        kernel = np.ones((5,5), np.uint8)\n        \n        # The first parameter is the original image, \n        # kernel is the matrix with which image is  \n        # convolved and third parameter is the number  \n        # of iterations, which will determine how much  \n        # you want to erode/dilate a given image.  \n        img_erosion = cv2.erode(image, kernel, iterations=1) \n        \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(img_erosion, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n    plt.suptitle(title)\n\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:15:41.514294Z","iopub.execute_input":"2022-08-11T08:15:41.514729Z","iopub.status.idle":"2022-08-11T08:15:51.851710Z","shell.execute_reply.started":"2022-08-11T08:15:41.514684Z","shell.execute_reply":"2022-08-11T08:15:51.850473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Dilation**\nThere! Now we can try dilation.","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title = '', aug = None):\n    width = 6\n    height = 4\n    fig, axs = plt.subplots(height, width, figsize=(15,15),constrained_layout = True)\n    for im in range(0, height * width):  \n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        kernel = np.ones((5,5), np.uint8) \n        \n        #dilation is performed here.\n        img_dilation = cv2.dilate(image, kernel, iterations=1) \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(img_dilation, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n        \n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's Erosion\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:15:51.853669Z","iopub.execute_input":"2022-08-11T08:15:51.854368Z","iopub.status.idle":"2022-08-11T08:16:01.815700Z","shell.execute_reply.started":"2022-08-11T08:15:51.854298Z","shell.execute_reply":"2022-08-11T08:16:01.814623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Combination of Erosion and Dilation**","metadata":{}},{"cell_type":"code","source":"def view_images_aug(images, title = '', aug = None):\n    width = 6\n    height = 4\n    fig, axs = plt.subplots(height, width, figsize=(15,15),constrained_layout = True)\n    for im in range(0, height * width):  \n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        kernel = np.ones((5,5), np.uint8) \n        #   \n        img_erosion = cv2.erode(image, kernel, iterations=1) \n        img_dilation = cv2.dilate(img_erosion, kernel, iterations=1) \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(img_dilation, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n        \n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's Erosion\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:16:01.817204Z","iopub.execute_input":"2022-08-11T08:16:01.818345Z","iopub.status.idle":"2022-08-11T08:16:11.953844Z","shell.execute_reply.started":"2022-08-11T08:16:01.818283Z","shell.execute_reply":"2022-08-11T08:16:11.952515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Complex Wavelet Transform**\n**Horizontal Detail**","metadata":{}},{"cell_type":"code","source":"import pywt\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n    \ndef circle_crop(img, sigmaX=10):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    \n    img = crop_image_from_gray(img)    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    img=cv2.addWeighted ( img,4, cv2.GaussianBlur( img , (0,0) , sigmaX) ,-4 ,128)\n    return img \n\ndef view_images_aug(images, title = '', aug = None):\n    width = 6\n    height = 5\n    fig, axs = plt.subplots(height, width, figsize=(15,15), constrained_layout=True)\n    for im in range(0, height * width):  \n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image= circle_crop(image) # this will create circular crop around image center\n        coeffs = pywt.dwt2(image, 'bior1.3') # wavelet(biorthogonal) transform \n        IM1, (IM2, IM3, IM4) = coeffs \n        image = IM2 # Horizontal Details \n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image) \n        axs[i,j].axis('off')\n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:16:11.955304Z","iopub.execute_input":"2022-08-11T08:16:11.955894Z","iopub.status.idle":"2022-08-11T08:16:25.119125Z","shell.execute_reply.started":"2022-08-11T08:16:11.955858Z","shell.execute_reply":"2022-08-11T08:16:25.117569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pywt\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n    \ndef circle_crop(img, sigmaX=10):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    \n    img = crop_image_from_gray(img)    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    img=cv2.addWeighted ( img,4, cv2.GaussianBlur( img , (0,0) , sigmaX) ,-4 ,128)\n    return img \n\ndef view_images_aug(images, title = '', aug = None):\n    width = 6\n    height = 5\n    fig, axs = plt.subplots(height, width, figsize=(15,15), constrained_layout=True)\n    for im in range(0, height * width):  \n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image= circle_crop(image) # # this will create circular crop around image center\n        coeffs = pywt.dwt2(image, 'bior1.3') # wavelet(biorthogonal) transform \n        IM1, (IM2, IM3, IM4) = coeffs\n        image = IM3 #vertical details\n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:16:25.121370Z","iopub.execute_input":"2022-08-11T08:16:25.122057Z","iopub.status.idle":"2022-08-11T08:16:39.478563Z","shell.execute_reply.started":"2022-08-11T08:16:25.122018Z","shell.execute_reply":"2022-08-11T08:16:39.477196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Hough Transform**","metadata":{}},{"cell_type":"code","source":"# def view_images_aug(images, title = '', aug = None):\n#     width = 6\n#     height = 4\n#     fig, axs = plt.subplots(height, width, figsize=(15,15), constrained_layout=True)\n#     for im in range(0, height * width):  \n#         data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm'))\n#         image = data.pixel_array\n#         image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n#         image = cv2.resize(image, (256, 256))\n        \n#         #apply hough transform on the circle cropped image\n#         crop_image= circle_crop(image)\n# #         crop_image = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n#         crop_image = cv2.normalize(src=crop_image, dst=None, alpha=0, beta=255, norm_type=cv2.NORM_MINMAX, dtype=cv2.CV_8U)\n#         crop_image = cv2.cvtColor(crop_image, cv2.COLOR_RGB2GRAY)    \n#     #         print(crop_image.shape)\n#         circles = cv2.HoughCircles(crop_image,cv2.HOUGH_GRADIENT,1,20,param1=50,param2=30,minRadius=0,maxRadius=0)\n# #         print(circles)\n# #         print(f\"Circles are : {circles.dtype}\")\n#         image = np.uint16(np.around(circles))\n#         i = im // width\n#         j = im % width\n#         axs[i,j].imshow(image, cmap=plt.cm.gray) \n#         axs[i,j].axis('off')\n#     plt.suptitle(title)\n# view_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:16:39.480214Z","iopub.execute_input":"2022-08-11T08:16:39.480621Z","iopub.status.idle":"2022-08-11T08:16:39.486073Z","shell.execute_reply.started":"2022-08-11T08:16:39.480584Z","shell.execute_reply":"2022-08-11T08:16:39.485159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def view_images_aug(images, title = '', aug = None):\n    width = 6\n    height = 4\n    fig, axs = plt.subplots(height, width, figsize=(15,15), constrained_layout=True)\n    for im in range(0, height * width):  \n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image= circle_crop(image)\n        gray = cv2.cvtColor(img,cv2.COLOR_BGR2GRAY)\n        edges = cv2.Canny(gray,50,150)\n\n#         lines = cv2.HoughLines(edges,1,np.pi/180,200)\n#         for rho,theta in lines[0]:\n#             a = np.cos(theta)\n#             b = np.sin(theta)\n#             x0 = a*rho\n#             y0 = b*rho\n#             x1 = int(x0 + 1000*(-b))\n#             y1 = int(y0 + 1000*(a))\n#             x2 = int(x0 - 1000*(-b))\n#             y2 = int(y0 - 1000*(a))\n\n#         image = cv2.line(img,(x1,y1),(x2,y2),(0,0,255),2)\n        i = im // width\n        j = im % width\n        axs[i,j].imshow(edges, cmap=plt.cm.gray) \n        axs[i,j].axis('off')\n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's growth\");","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:16:39.487516Z","iopub.execute_input":"2022-08-11T08:16:39.488822Z","iopub.status.idle":"2022-08-11T08:17:10.075333Z","shell.execute_reply.started":"2022-08-11T08:16:39.488746Z","shell.execute_reply":"2022-08-11T08:17:10.073476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Adding Salt and Pepper Noise**","metadata":{}},{"cell_type":"code","source":"import random\ndef sp_noise(image,prob):\n    '''\n    Add salt and pepper noise to image\n    prob: Probability of the noise\n    '''\n    output = np.zeros(image.shape,np.uint8)\n    thres = 1 - prob \n    for i in range(image.shape[0]):\n        for j in range(image.shape[1]):\n            rdn = random.random()\n            if rdn < prob:\n                output[i][j] = 0\n            elif rdn > thres:\n                output[i][j] = 255\n            else:\n                output[i][j] = image[i][j]\n    return output\n\ndef view_images_aug(images, title = '', aug = None):\n    width = 6\n    height = 5\n    fig, axs = plt.subplots(height, width, figsize=(15,15))\n    for im in range(0, height * width):  \n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image = sp_noise(image, 0.192)\n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image, cmap=plt.cm.bone) \n        axs[i,j].axis('off')\n        \n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's Erosion\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:17:10.077831Z","iopub.execute_input":"2022-08-11T08:17:10.078243Z","iopub.status.idle":"2022-08-11T08:17:24.070770Z","shell.execute_reply.started":"2022-08-11T08:17:10.078207Z","shell.execute_reply":"2022-08-11T08:17:24.069240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Cutout Data Augmentation**\nCutout is a simple regularization technique for convolutional neural networks that involves removing contiguous sections of input images, effectively augmenting the dataset with partially occluded versions of existing samples.","metadata":{}},{"cell_type":"code","source":"def get_random_eraser(input_img,p=0.5, s_l=0.02, s_h=0.4, r_1=0.3, r_2=1/0.3, v_l=0, v_h=255, pixel_level=False):\n   # def eraser(input_img):\n    img_h, img_w, img_c = input_img.shape\n\n    while True:\n        s = np.random.uniform(s_l, s_h) * img_h * img_w\n        r = np.random.uniform(r_1, r_2)\n        w = int(np.sqrt(s / r))\n        h = int(np.sqrt(s * r))\n        left = np.random.randint(0, img_w)\n        top = np.random.randint(0, img_h)\n\n        if left + w <= img_w and top + h <= img_h:\n            break\n\n    if pixel_level:\n        c = np.random.uniform(v_l, v_h, (h, w, img_c))\n    else:\n        c = np.random.uniform(v_l, v_h)\n\n    input_img[top:top + h, left:left + w, :] = c\n\n    return input_img\n\ndef view_images_aug(images, title = '', aug = None):\n    width = 6\n    height = 4\n    fig, axs = plt.subplots(height, width, figsize=(15,15), constrained_layout = True)\n    for im in range(0, height * width):  \n        data = pydicom.read_file(os.path.join(train_images_dir, list(images)[im]+ '.dcm'))\n        image = data.pixel_array\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = get_random_eraser(image)\n#         image = cv2.resize(image, (256, 256))\n#         image = sp_noise(image, 0.192)\n        i = im // width\n        j = im % width\n        axs[i,j].imshow(image) \n        axs[i,j].axis('off')\n        \n    plt.suptitle(title)\nview_images_aug(train_df_1[train_df_1['diagnosis']=='lentigo NOS']['image_name'], title=\"Lentigo NOS's Cutout\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:39:10.426386Z","iopub.execute_input":"2022-08-11T08:39:10.428111Z","iopub.status.idle":"2022-08-11T08:39:46.481273Z","shell.execute_reply.started":"2022-08-11T08:39:10.428034Z","shell.execute_reply":"2022-08-11T08:39:46.480291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# image = cv2.imread(\"../input/siim-isic-melanoma-classification/jpeg/train/ISIC_0015719.jpg\")\n# gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n# print(gray)\n# tol = 7\n# mask = gray > tol #mask => boolean array\n# print(mask)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:18:06.734167Z","iopub.execute_input":"2022-08-11T08:18:06.735142Z","iopub.status.idle":"2022-08-11T08:18:07.146069Z","shell.execute_reply.started":"2022-08-11T08:18:06.735092Z","shell.execute_reply":"2022-08-11T08:18:07.144420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mask.any(1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:18:07.147586Z","iopub.execute_input":"2022-08-11T08:18:07.148211Z","iopub.status.idle":"2022-08-11T08:18:07.155946Z","shell.execute_reply.started":"2022-08-11T08:18:07.148175Z","shell.execute_reply":"2022-08-11T08:18:07.154833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mask.any(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:18:07.157426Z","iopub.execute_input":"2022-08-11T08:18:07.157737Z","iopub.status.idle":"2022-08-11T08:18:07.173567Z","shell.execute_reply.started":"2022-08-11T08:18:07.157705Z","shell.execute_reply":"2022-08-11T08:18:07.171095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# my_2d_array = np.arange(start = 1, stop = 7).reshape((2,3))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:18:07.175815Z","iopub.execute_input":"2022-08-11T08:18:07.176535Z","iopub.status.idle":"2022-08-11T08:18:07.181952Z","shell.execute_reply.started":"2022-08-11T08:18:07.176494Z","shell.execute_reply":"2022-08-11T08:18:07.180541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# my_2d_array","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:18:07.183632Z","iopub.execute_input":"2022-08-11T08:18:07.184003Z","iopub.status.idle":"2022-08-11T08:18:07.197175Z","shell.execute_reply.started":"2022-08-11T08:18:07.183970Z","shell.execute_reply":"2022-08-11T08:18:07.196158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(my_2d_array.any(0))\n# print(my_2d_array.any(1))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:18:07.198726Z","iopub.execute_input":"2022-08-11T08:18:07.199387Z","iopub.status.idle":"2022-08-11T08:18:07.207789Z","shell.execute_reply.started":"2022-08-11T08:18:07.199345Z","shell.execute_reply":"2022-08-11T08:18:07.206800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.ix_(my_2d_array.any(1),my_2d_array.any(0))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:18:07.209213Z","iopub.execute_input":"2022-08-11T08:18:07.210598Z","iopub.status.idle":"2022-08-11T08:18:07.222766Z","shell.execute_reply.started":"2022-08-11T08:18:07.210559Z","shell.execute_reply":"2022-08-11T08:18:07.221092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# my_2d_array[np.ix_(my_2d_array.any(1),my_2d_array.any(0))]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:18:07.225166Z","iopub.execute_input":"2022-08-11T08:18:07.226040Z","iopub.status.idle":"2022-08-11T08:18:07.235145Z","shell.execute_reply.started":"2022-08-11T08:18:07.225992Z","shell.execute_reply":"2022-08-11T08:18:07.233772Z"},"trusted":true},"execution_count":null,"outputs":[]}]}