{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div align='center'><font size=\"5\" color='#353B47'>SIIM ISIC</font></div>\n<div align='center'><font size=\"4\" color=\"#353B47\">Exploratory Data Analysis</font></div>\n<br>\n<hr>","metadata":{}},{"cell_type":"markdown","source":"# 1. Load libraries and data","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport random\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport os\nfrom sklearn.utils import resample\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly import tools\nfrom plotly.offline import iplot, init_notebook_mode\n\ninit_notebook_mode()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-06-04T21:08:38.310433Z","iopub.execute_input":"2021-06-04T21:08:38.310796Z","iopub.status.idle":"2021-06-04T21:08:40.657167Z","shell.execute_reply.started":"2021-06-04T21:08:38.310766Z","shell.execute_reply":"2021-06-04T21:08:40.655907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_to_images = '../input/siim-isic-melanoma-classification/jpeg/'\nPATH_to_dataframes = '../input/siim-isic-melanoma-classification/'","metadata":{"execution":{"iopub.status.busy":"2021-06-04T21:08:40.658609Z","iopub.execute_input":"2021-06-04T21:08:40.659064Z","iopub.status.idle":"2021-06-04T21:08:40.664378Z","shell.execute_reply.started":"2021-06-04T21:08:40.659030Z","shell.execute_reply":"2021-06-04T21:08:40.663341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import data\ntrain = pd.read_csv(PATH_to_dataframes + \"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-06-04T21:08:40.665749Z","iopub.execute_input":"2021-06-04T21:08:40.666048Z","iopub.status.idle":"2021-06-04T21:08:40.762455Z","shell.execute_reply.started":"2021-06-04T21:08:40.666018Z","shell.execute_reply":"2021-06-04T21:08:40.761439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Clean data","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'{train.shape[0]} observations, {train.shape[1]} columns')","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing values per column\ntrain.isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop missing values\ntrain = train.dropna()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T21:08:40.763795Z","iopub.execute_input":"2021-06-04T21:08:40.764117Z","iopub.status.idle":"2021-06-04T21:08:40.798264Z","shell.execute_reply.started":"2021-06-04T21:08:40.764074Z","shell.execute_reply":"2021-06-04T21:08:40.797016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. EDA","metadata":{}},{"cell_type":"code","source":"# Display proportion of benign and malignant melanomas\ntrain.benign_malignant.value_counts(normalize = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Facing a imbalanced classification problem.","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(train, x=\"benign_malignant\",\n                   hover_data=train.columns)\nfig.update_layout(title_text='Count of benign/malignant')\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(train, x=\"anatom_site_general_challenge\",\n                   hover_data=train.columns)\nfig.update_layout(title_text='Anatom sites')\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Categorizing age\ntrain['age_cat'] = '0 / 20 years'\ntrain.loc[(train['age_approx'] > 20) & (train['age_approx'] <= 40), 'age_cat'] = '20 / 40 years'\ntrain.loc[(train['age_approx'] > 40) & (train['age_approx'] <= 60), 'age_cat'] = '40 / 60 years'\ntrain.loc[(train['age_approx'] > 60) & (train['age_approx'] <= 80), 'age_cat'] = '60 / 80 years'\ntrain.loc[(train['age_approx'] > 80), 'age_approx'] = '80+ years'","metadata":{"execution":{"iopub.status.busy":"2021-06-04T21:08:40.799695Z","iopub.execute_input":"2021-06-04T21:08:40.799980Z","iopub.status.idle":"2021-06-04T21:08:40.833927Z","shell.execute_reply.started":"2021-06-04T21:08:40.799952Z","shell.execute_reply":"2021-06-04T21:08:40.832827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate minority and majority class\ntrain_no_target = train[train['target']==0]\ntrain_target = train[train['target']==1]","metadata":{"execution":{"iopub.status.busy":"2021-06-04T21:08:40.835666Z","iopub.execute_input":"2021-06-04T21:08:40.835965Z","iopub.status.idle":"2021-06-04T21:08:40.850145Z","shell.execute_reply.started":"2021-06-04T21:08:40.835938Z","shell.execute_reply":"2021-06-04T21:08:40.849237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image(df):\n    \n    random_sampling = [random.randint(0, len(df)) for i in range(9)]\n    image_indexes = [list(df.index)[random_sampling[i]] for i in range(len(random_sampling))]\n    \n    i = 0\n    \n    # plot first few images\n    plt.figure(figsize=(12,12))\n    for index in image_indexes:\n        \n        # Get corresponding label\n        image_name = df.loc[index, 'image_name']\n        site = df.loc[index, \"anatom_site_general_challenge\"]\n        target = df.loc[index, \"target\"]        \n        \n        # define subplot\n        plt.subplot(330 + 1 + i)\n        plt.title('Target: %s \\n'%target+\\\n                  'Site: %s\\n'%site,\n                  fontsize=18)\n        \n        # plot raw pixel data\n        numpy_image = cv2.imread(PATH_to_images + \"train/\" + image_name + \".jpg\")\n        plt.imshow(cv2.cvtColor(numpy_image, cv2.COLOR_BGR2RGB))\n        i+=1\n        \n    plt.subplots_adjust(bottom = 0.001)  # the bottom of the subplots of the figure\n    plt.subplots_adjust(top = 0.99)\n    # show the figure\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T21:13:02.924132Z","iopub.execute_input":"2021-06-04T21:13:02.924571Z","iopub.status.idle":"2021-06-04T21:13:02.938804Z","shell.execute_reply.started":"2021-06-04T21:13:02.924539Z","shell.execute_reply":"2021-06-04T21:13:02.937371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image(train_no_target)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T21:13:03.196541Z","iopub.execute_input":"2021-06-04T21:13:03.196885Z","iopub.status.idle":"2021-06-04T21:13:13.366530Z","shell.execute_reply.started":"2021-06-04T21:13:03.196856Z","shell.execute_reply":"2021-06-04T21:13:13.364982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image(train_target)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I would like to represent the target through all columns in the dataframe. To do so, I will use a parallel category diagram.","metadata":{}},{"cell_type":"code","source":"train_parallel = train[['image_name', 'patient_id', 'sex', 'age_cat', 'anatom_site_general_challenge', 'diagnosis', 'target']]\n\nfig = px.parallel_categories(train_parallel, color=\"target\", color_continuous_scale=px.colors.sequential.algae)\nfig.update_layout(title='Parallel category diagram on trainset')\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The problem here is that it is barely readable as our data are imbalanced. I will perform downsampling on the data and check the changes.","metadata":{}},{"cell_type":"code","source":"train.target.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Downsampling majority class\ndf_majority_downsampled = resample(train_no_target, \n                                   replace=False, # sample without replacement\n                                   n_samples=584, # to match minority class\n                                   random_state=42)\n \n# Combine minority class with downsampled majority class\ntrain_downsampled = pd.concat([df_majority_downsampled, train_target])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_parallel_downsampled = train_downsampled[['image_name', 'patient_id', 'sex', 'age_cat', 'anatom_site_general_challenge', 'diagnosis', 'target']]\n\nfig = px.parallel_categories(train_parallel_downsampled, color=\"target\", color_continuous_scale=px.colors.sequential.algae)\nfig.update_layout(title='Parallel category diagram on downsampled trainset')\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This visualization is also quite useful to check undesirable categories in categorical columns that don't count as missing value. Here I can see that I didn't clean the sex column.","metadata":{}},{"cell_type":"markdown","source":"<hr>\n<br>\n<div align='justify'><font color=\"#353B47\" size=\"4\">Thank you for taking the time to read this notebook. I hope that I was able to answer your questions or your curiosity and that it was quite understandable. <u>any constructive comments are welcome</u>. They help me progress and motivate me to share better quality content. I am above all a passionate person who tries to advance my knowledge but also that of others. If you liked it, feel free to <u>upvote and share my work.</u> </font></div>\n<br>\n<div align='center'><font color=\"#353B47\" size=\"3\">Thank you and may passion guide you.</font></div>","metadata":{}}]}