{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src='https://i.imgur.com/odXiwBt.png'>\n<h1><center>🧬 Melanoma Classification🧬: EDA + Augmentations</center><h1>\n\n<img src='https://i.imgur.com/Jxtc8x0.png' width=500>\n\n# 1. Introduction ▶\n\n### 1.1 What is Melanoma? Stats and Facts:\n* [Melanoma is the least common but the most deadly skin cancer, accounting for only about 1% of all cases, but the vast majority of skin cancer death.](https://www.aimatmelanoma.org/about-melanoma/melanoma-stats-facts-and-figures/)\n* Melanoma is the third most common cancer among men and women ages 20-39.\n* In the U.S., melanoma continues to be \n    * the fifth most common cancer in men of all age groups\n    * the sixth most common cancer in women of all age groups\n* The world’s highest incidence of melanoma is in Australia and New Zealand (more than twice as high as in North America)\n\n\n### 1.2 What we need to do? Data and Overview:\n> The purpose is to correctly identify the **benign** and **malignant** cases. A *benign* tumor is a tumor that DOES NOT invade its surrounding tissue or spread around the body. A *malignant* tumor is a tumor that MAY invade its surrounding tissue or spread around the body.\n<img src = 'https://www.verywellhealth.com/thmb/IFgBpbmhYCJdS4rvLACzX3Ukqsc=/1500x0/filters:no_upscale():max_bytes(150000):strip_icc():format(webp)/514240-article-img-malignant-vs-benign-tumor2111891f-54cc-47aa-8967-4cd5411fdb2f-5a2848f122fa3a0037c544be.png' width = 300>\n\n> Data: DICOM Files split in Train (33,126 observations) and Test (10,982 observations)\n<img src='https://i.imgur.com/or0AoVs.png' width = 500>\n\n### 1.3 Metrics of Evaluation. Area under the ROC curve:\n* [The ROC curve is created by plotting the true positive rate (TPR) against the false positive rate (FPR) at various threshold settings.](https://en.wikipedia.org/wiki/Receiver_operating_characteristic)\n\n# 2. Libraries 📚","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-03T14:41:49.057536Z","iopub.execute_input":"2024-12-03T14:41:49.057974Z","iopub.status.idle":"2024-12-03T14:41:49.097169Z","shell.execute_reply.started":"2024-12-03T14:41:49.057933Z","shell.execute_reply":"2024-12-03T14:41:49.095252Z"}}},{"cell_type":"code","source":"# Regular Imports\nimport os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'  # Suppress TensorFlow info and warnings\nimport pandas as pd\nimport numpy as np\nimport shutil\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.image as mpimg\nfrom tabulate import tabulate\nimport missingno as msno\nfrom IPython.display import display_html\nfrom PIL import Image\nimport gc\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import classification_report\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\n# TensorFlow & Keras\nfrom tensorflow.keras.applications import EfficientNetB0, VGG16\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Input, Dense, GlobalAveragePooling2D, Dropout, concatenate\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras import layers, models  # Correctly importing layers and models\n\n# For DICOM\nimport pydicom  # for DICOM images\nfrom skimage.transform import resize\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Set Color Palettes for the notebook\ncolors_nude = ['#e0798c','#65365a','#da8886','#cfc4c4','#dfd7ca']\nsns.palplot(sns.color_palette(colors_nude))\n\n# Set Style\nsns.set_style(\"whitegrid\")\nsns.despine(left=True, bottom=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:03:31.735387Z","iopub.execute_input":"2024-12-06T17:03:31.736093Z","iopub.status.idle":"2024-12-06T17:03:31.820296Z","shell.execute_reply.started":"2024-12-06T17:03:31.736058Z","shell.execute_reply":"2024-12-06T17:03:31.819043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list(os.listdir('../input/siim-isic-melanoma-classification'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:21.070459Z","iopub.execute_input":"2024-12-06T16:30:21.070817Z","iopub.status.idle":"2024-12-06T16:30:21.076707Z","shell.execute_reply.started":"2024-12-06T16:30:21.070786Z","shell.execute_reply":"2024-12-06T16:30:21.075877Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. CSV Files - Train📁 + Test📂","metadata":{}},{"cell_type":"code","source":"# Directory\ndirectory = '../input/siim-isic-melanoma-classification'\n\n# Import the 2 csv s\ntrain_df = pd.read_csv(directory + '/train.csv')\ntest_df = pd.read_csv(directory + '/test.csv')\n\nprint('Train has {:,} rows and Test has {:,} rows.'.format(len(train_df), len(test_df)))\n\n# Change columns names\nnew_names = ['dcm_name', 'ID', 'sex', 'age', 'anatomy', 'diagnosis', 'benign_malignant', 'target']\ntrain_df.columns = new_names\ntest_df.columns = new_names[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:23.387005Z","iopub.execute_input":"2024-12-06T16:30:23.387627Z","iopub.status.idle":"2024-12-06T16:30:23.491941Z","shell.execute_reply.started":"2024-12-06T16:30:23.387592Z","shell.execute_reply":"2024-12-06T16:30:23.491075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:14.614501Z","iopub.execute_input":"2024-12-06T10:56:14.614766Z","iopub.status.idle":"2024-12-06T10:56:14.626142Z","shell.execute_reply.started":"2024-12-06T10:56:14.614741Z","shell.execute_reply":"2024-12-06T10:56:14.625246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:27.060050Z","iopub.execute_input":"2024-12-06T16:30:27.060381Z","iopub.status.idle":"2024-12-06T16:30:27.076221Z","shell.execute_reply.started":"2024-12-06T16:30:27.060351Z","shell.execute_reply":"2024-12-06T16:30:27.075282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1_styler = train_df.head().style.set_table_attributes(\"style='display:inline'\").set_caption('Head Train Data')\ndf2_styler = test_df.head().style.set_table_attributes(\"style='display:inline'\").set_caption('Head Test Data')\n\ndisplay_html(df1_styler._repr_html_() + df2_styler._repr_html_(), raw=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:32.579797Z","iopub.execute_input":"2024-12-06T16:30:32.580133Z","iopub.status.idle":"2024-12-06T16:30:32.644810Z","shell.execute_reply.started":"2024-12-06T16:30:32.580105Z","shell.execute_reply":"2024-12-06T16:30:32.644000Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.1 Missing Values ❓\r\n\r\nLet's first visualize the missing values.","metadata":{}},{"cell_type":"code","source":"f, (ax1, ax2) = plt.subplots(1, 2, figsize = (16, 6))\n\nmsno.matrix(train_df, ax = ax1, color=(207/255, 196/255, 171/255), fontsize=10)\nmsno.matrix(test_df, ax = ax2, color=(218/255, 136/255, 130/255), fontsize=10)\n\nax1.set_title('Train Missing Values Map', fontsize = 16)\nax2.set_title('Test Missing Values Map', fontsize = 16);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:35.594805Z","iopub.execute_input":"2024-12-06T16:30:35.595820Z","iopub.status.idle":"2024-12-06T16:30:36.244216Z","shell.execute_reply.started":"2024-12-06T16:30:35.595784Z","shell.execute_reply":"2024-12-06T16:30:36.243290Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train**:\n1. `sex`: 65 missing values (0.2% of total data)\n2. `age`: 68 missing values (correspond with `sex` missingness)\n3. `anatomy`: 527 missing values (1.59% of total data)\n\n**Test**:\n1. `anatomy`: 351 missing values (3.1% of total data)\n\nLet's take them 1 by 1 and deal with em.\n\n### Train: SEX Variable\n\n> All missing values are *benign* and the majority of the patients have the Melanoma in the Lower Extremity, Upper Extremity and Torso. All values for `diagnosis` are unknown. Therefore, we'll use the most predominant gender that appears in these values to impute the missing values.","metadata":{"execution":{"iopub.status.busy":"2024-12-03T14:52:47.802935Z","iopub.execute_input":"2024-12-03T14:52:47.803308Z","iopub.status.idle":"2024-12-03T14:52:47.811796Z","shell.execute_reply.started":"2024-12-03T14:52:47.803275Z","shell.execute_reply":"2024-12-03T14:52:47.810385Z"}}},{"cell_type":"code","source":"# Separate Data\nnan_sex = train_df[train_df['sex'].isna()]\nis_sex = train_df[~train_df['sex'].isna()]\n\n# Plotting\nf, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\nsns.countplot(data=nan_sex, x='anatomy', ax=ax1, palette=colors_nude)\nsns.countplot(data=is_sex, x='anatomy', ax=ax2, palette=colors_nude)\n\nax1.set_title('NAN Gender: Anatomy', fontsize=16)\nax2.set_title('Rest Gender: Anatomy', fontsize=16)\n\nfor ax in [ax1, ax2]:\n    ax.set_xticklabels(ax.get_xticklabels(), rotation=35, ha=\"right\")\nsns.despine(left=True, bottom=True)\n\n# Benign/Malignant Check\nbenign_count = nan_sex['benign_malignant'].value_counts().get('benign', 0)\nmalignant_count = nan_sex['benign_malignant'].value_counts().get('malignant', 0)\ntotal_nan = nan_sex.shape[0]\n\nprint(f'Out of {total_nan} NAN values, {benign_count} are benign and {malignant_count} are malignant.')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:37.683064Z","iopub.execute_input":"2024-12-06T16:30:37.684084Z","iopub.status.idle":"2024-12-06T16:30:38.215272Z","shell.execute_reply.started":"2024-12-06T16:30:37.684045Z","shell.execute_reply":"2024-12-06T16:30:38.214376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check how many are males and how many females\nanatomy = ['lower extremity', 'upper extremity', 'torso']\ntrain_df[(train_df['anatomy'].isin(anatomy)) & (train_df['target'] == 0)]['sex'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:41.239211Z","iopub.execute_input":"2024-12-06T16:30:41.239581Z","iopub.status.idle":"2024-12-06T16:30:41.255841Z","shell.execute_reply.started":"2024-12-06T16:30:41.239548Z","shell.execute_reply":"2024-12-06T16:30:41.254839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute the missing values with male\ntrain_df['sex'].fillna(\"male\", inplace = True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:43.170357Z","iopub.execute_input":"2024-12-06T16:30:43.171151Z","iopub.status.idle":"2024-12-06T16:30:43.179304Z","shell.execute_reply.started":"2024-12-06T16:30:43.171101Z","shell.execute_reply":"2024-12-06T16:30:43.178453Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train: AGE Variable\r\n> The distributions and values are very similar with the missingness patern in `sex` variable. So, we'll impute in the same manner. The *mean* and *median* of `age` variable has the same value of 50, while the *mode* is at 45. The distribution is normal, so we'll use the MEDIAN to impute.","metadata":{}},{"cell_type":"code","source":"# Separate Data\nnan_age = train_df[train_df['age'].isna()]\nis_age = train_df[~train_df['age'].isna()]\n\n# Plotting\nf, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\nsns.countplot(data=nan_age, x='anatomy', ax=ax1, palette=colors_nude)\nsns.countplot(data=is_age, x='anatomy', ax=ax2, palette=colors_nude)\n\nax1.set_title('NAN Age: Anatomy', fontsize=16)\nax2.set_title('Rest Age: Anatomy', fontsize=16)\n\nfor ax in [ax1, ax2]:\n    ax.set_xticklabels(ax.get_xticklabels(), rotation=35, ha=\"right\")\nsns.despine(left=True, bottom=True)\n\n# Benign/Malignant Check\nbenign_count = nan_age['benign_malignant'].value_counts().get('benign', 0)\nmalignant_count = nan_age['benign_malignant'].value_counts().get('malignant', 0)\ntotal_nan = nan_age.shape[0]\n\nprint(f'Out of {total_nan} NAN values, {benign_count} are benign and {malignant_count} are malignant.')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:15.780810Z","iopub.execute_input":"2024-12-06T10:56:15.781044Z","iopub.status.idle":"2024-12-06T10:56:16.295932Z","shell.execute_reply.started":"2024-12-06T10:56:15.781021Z","shell.execute_reply":"2024-12-06T10:56:16.295036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the mode age\nanatomy = ['lower extremity', 'upper extremity', 'torso']\nmode = train_df[(train_df['anatomy'].isin(anatomy)) & (train_df['target'] == 0) & (train_df['sex'] == 'male')]['age'].mode()\nprint('Mode is:', mode)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:47.530723Z","iopub.execute_input":"2024-12-06T16:30:47.531087Z","iopub.status.idle":"2024-12-06T16:30:47.546341Z","shell.execute_reply.started":"2024-12-06T16:30:47.531059Z","shell.execute_reply":"2024-12-06T16:30:47.545531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute the missing values with Mode\ntrain_df['age'].fillna(mode, inplace = True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:49.373324Z","iopub.execute_input":"2024-12-06T16:30:49.373694Z","iopub.status.idle":"2024-12-06T16:30:49.381116Z","shell.execute_reply.started":"2024-12-06T16:30:49.373658Z","shell.execute_reply":"2024-12-06T16:30:49.380226Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train: ANATOMY Variable\r\n> First, we need to keep in mind that between the missing data there are 9 malignant cases, so we should treat the imputation separate for both benign and malignant. In terms of `age` and `gender`, both missing and not missing data seem to behave about the same. However, the most frequent anatomy for both benign and malignant is *torso*, so we'll impute this value.","metadata":{}},{"cell_type":"code","source":"# Add flag for missing anatomy\nanatomy = train_df.copy()\nanatomy['flag'] = np.where(anatomy['anatomy'].isna(), 'missing', 'not_missing')\n\n# Figure\nf, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n# Countplot for Gender and Missing Anatomy\nsns.countplot(x='flag', hue='sex', data=anatomy, ax=ax1, palette=colors_nude)\nax1.set_title('Gender for Anatomy', fontsize=16)\n\n# KDE Plot for Age Distribution\nsns.kdeplot(data=anatomy[anatomy['flag'] == 'missing'], x='age', label='Missing',\n            ax=ax2, color=colors_nude[2], linewidth=4)\nsns.kdeplot(data=anatomy[anatomy['flag'] == 'not_missing'], x='age', label='Not Missing',\n            ax=ax2, color=colors_nude[3], linewidth=4)\nax2.set_title('Age Distribution for Anatomy', fontsize=16)\n\nsns.despine(left=True, bottom=True)\nax2.legend(title='Flag')\n\n# Benign/Malignant Check\nben_mal = anatomy[anatomy['flag'] == 'missing']['benign_malignant'].value_counts()\nbenign_count = ben_mal.get('benign', 0)\nmalignant_count = ben_mal.get('malignant', 0)\n\nprint(f'From all missing values, {benign_count} are benign and {malignant_count} malignant.')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:16.320293Z","iopub.execute_input":"2024-12-06T10:56:16.320546Z","iopub.status.idle":"2024-12-06T10:56:17.046802Z","shell.execute_reply.started":"2024-12-06T10:56:16.320523Z","shell.execute_reply":"2024-12-06T10:56:17.045943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute for anatomy\ntrain_df['anatomy'].fillna('torso', inplace = True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:54.833654Z","iopub.execute_input":"2024-12-06T16:30:54.834345Z","iopub.status.idle":"2024-12-06T16:30:54.840643Z","shell.execute_reply.started":"2024-12-06T16:30:54.834312Z","shell.execute_reply":"2024-12-06T16:30:54.839808Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Test: ANATOMY Variable\r\n> The majority of the people with missing `anatomy` have 70 yo, so we'll use the anatomy with the biggest frequency for age 70.","metadata":{}},{"cell_type":"code","source":"# Add flag for missing anatomy\nanatomy = test_df.copy()\nanatomy['flag'] = np.where(anatomy['anatomy'].isna(), 'missing', 'not_missing')\n\n# Figure\nf, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n# Corrected countplot\nsns.countplot(x='flag', hue='sex', data=anatomy, ax=ax1, palette=colors_nude)\nax1.set_title('Gender for Anatomy', fontsize=16)\n\n# Replace sns.distplot with sns.kdeplot (since distplot is deprecated)\nsns.kdeplot(data=anatomy[anatomy['flag'] == 'missing'], x='age', \n            label='Missing', ax=ax2, color=colors_nude[2], linewidth=4, bw_adjust=0.1)\nsns.kdeplot(data=anatomy[anatomy['flag'] == 'not_missing'], x='age', \n            label='Not Missing', ax=ax2, color=colors_nude[3], linewidth=4, bw_adjust=0.1)\nax2.set_title('Age Distribution for Anatomy', fontsize=16)\nax2.legend(title='Flag')\n\nsns.despine(left=True, bottom=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:17.056144Z","iopub.execute_input":"2024-12-06T10:56:17.056454Z","iopub.status.idle":"2024-12-06T10:56:17.715854Z","shell.execute_reply.started":"2024-12-06T10:56:17.056430Z","shell.execute_reply":"2024-12-06T10:56:17.715142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"anatomy_counts = test_df[test_df['age'] == 70]['anatomy'].value_counts().reset_index()\nanatomy_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:17.716780Z","iopub.execute_input":"2024-12-06T10:56:17.717014Z","iopub.status.idle":"2024-12-06T10:56:17.727774Z","shell.execute_reply.started":"2024-12-06T10:56:17.716992Z","shell.execute_reply":"2024-12-06T10:56:17.726993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute the value\ntest_df['anatomy'].fillna('torso', inplace = True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:30:58.642693Z","iopub.execute_input":"2024-12-06T16:30:58.643077Z","iopub.status.idle":"2024-12-06T16:30:58.648766Z","shell.execute_reply.started":"2024-12-06T16:30:58.643045Z","shell.execute_reply":"2024-12-06T16:30:58.647709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the files\ntrain_df.to_csv('train_clean.csv', index=False)\ntest_df.to_csv('test_clean.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:31:00.580627Z","iopub.execute_input":"2024-12-06T16:31:00.581655Z","iopub.status.idle":"2024-12-06T16:31:00.701590Z","shell.execute_reply.started":"2024-12-06T16:31:00.581610Z","shell.execute_reply":"2024-12-06T16:31:00.700647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.2 EDA - Let's take a look 🔎\r\n\r\n### Target Variable:\r\n1. Very HIGH class imbalance. We need to take this in consideration when Modeling.\r\n2. Age distribution:\r\n    * Benign: follows a normal distribution\r\n    * Malignant: a little skewed to the left, with the peak oriented towards higher age values.","metadata":{}},{"cell_type":"code","source":"# Figure setup\nf, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n# Count plot for benign and malignant cases\nsns.countplot(data=train_df, x='benign_malignant', palette=colors_nude[2:4], ax=ax1)\nfor p in ax1.patches:\n    ax1.annotate(\n        format(p.get_height(), ','),\n        (p.get_x() + p.get_width() / 2., p.get_height()),\n        ha='center', va='center', xytext=(0, 4), textcoords='offset points'\n    )\nax1.set_title('Benign vs Malignant Cases', fontsize=14)\n\n# Age distribution for benign and malignant cases\nsns.kdeplot(data=train_df[train_df['target'] == 0], x='age', ax=ax2, color=colors_nude[2], \n            linewidth=3, label='Benign')\nsns.kdeplot(data=train_df[train_df['target'] == 1], x='age', ax=ax2, color=colors_nude[3], \n            linewidth=3, label='Malignant')\nax2.set_title('Age Distribution by Target Type', fontsize=14)\nax2.legend()\n\nsns.despine(left=True, bottom=True)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:17.864089Z","iopub.execute_input":"2024-12-06T10:56:17.864454Z","iopub.status.idle":"2024-12-06T10:56:18.625556Z","shell.execute_reply.started":"2024-12-06T10:56:17.864413Z","shell.execute_reply":"2024-12-06T10:56:18.624679Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Target and Genders:\r\n1. There are more males than females in the dataset\r\n2. However, the percentages are ~ the same","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\na = sns.countplot(data=train_df, x='benign_malignant', hue='sex', palette=colors_nude)\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n\nplt.title('Gender split by Target Variable', fontsize=16)\nsns.despine(left=True, bottom=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:18.626804Z","iopub.execute_input":"2024-12-06T10:56:18.627478Z","iopub.status.idle":"2024-12-06T10:56:18.912081Z","shell.execute_reply.started":"2024-12-06T10:56:18.627430Z","shell.execute_reply":"2024-12-06T10:56:18.911287Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Anatomy and Target\r\n> Note: Distributions are about the same shape for both benign and malignant cases.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\na = sns.countplot(data=train_df, x='benign_malignant', hue='anatomy', palette=colors_nude)\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n\nplt.title('Anatomy split by Target Variable', fontsize=16)\nsns.despine(left=True, bottom=True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:18.913291Z","iopub.execute_input":"2024-12-06T10:56:18.913634Z","iopub.status.idle":"2024-12-06T10:56:19.294531Z","shell.execute_reply.started":"2024-12-06T10:56:18.913598Z","shell.execute_reply":"2024-12-06T10:56:19.293513Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Diagnosis and Target","metadata":{}},{"cell_type":"code","source":"# Filter data to exclude 'unknown'\nfiltered_df = train_df[train_df['diagnosis'] != 'unknown']\n\n# Create plots\nf, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n# Plot benign cases\na = sns.countplot(data=filtered_df[filtered_df['target'] == 0], x='diagnosis', ax=ax1, palette=colors_nude)\n\n# Plot malignant cases\nb = sns.countplot(data=filtered_df[filtered_df['target'] == 1], x='diagnosis', ax=ax2, palette=colors_nude)\n\n# Rotate x-axis labels for better visibility\na.set_xticklabels(a.get_xticklabels(), rotation=35, ha=\"right\")\nb.set_xticklabels(b.get_xticklabels(), rotation=35, ha=\"right\")\n\n# Annotate bar heights\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n               (p.get_x() + p.get_width() / 2., p.get_height()), \n               ha='center', va='center', xytext=(0, 4), textcoords='offset points')\n\nfor p in b.patches:\n    b.annotate(format(p.get_height(), ','), \n               (p.get_x() + p.get_width() / 2., p.get_height()), \n               ha='center', va='center', xytext=(0, 4), textcoords='offset points')\n\n# Set titles\nax1.set_title('Benign Cases: Diagnosis View', fontsize=16)\nax2.set_title('Malignant Cases: Diagnosis View', fontsize=16)\n\nsns.despine(left=True, bottom=True)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:19.295711Z","iopub.execute_input":"2024-12-06T10:56:19.295966Z","iopub.status.idle":"2024-12-06T10:56:19.701070Z","shell.execute_reply.started":"2024-12-06T10:56:19.295942Z","shell.execute_reply":"2024-12-06T10:56:19.700290Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Test Dataset Overview\r\n> Distributions look ~ the same as in Train Data.","metadata":{}},{"cell_type":"code","source":"# Calculate the counts for each gender\nsex_counts = train_df['sex'].value_counts()\n\n# Define sizes and labels\nsize = sex_counts.values\nlabels = sex_counts.index\nexplode = [0.1 if label == 'male' else 0 for label in labels]  # Explode male for emphasis (optional)\n\n# Plot\nplt.figure(figsize=(5, 5))\nplt.pie(size, labels=labels, explode=explode, shadow=True, startangle=90, colors=[\"r\", \"c\"], autopct='%1.1f%%')\nplt.title(\"Gender Distribution (Male vs Female)\", fontsize=18)\nplt.legend(title=\"Gender\")\nplt.show()\n\n# Print counts\nfor gender, count in zip(labels, size):\n    print(f\"{gender.capitalize()} Cases = {count}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:19.717541Z","iopub.execute_input":"2024-12-06T10:56:19.717907Z","iopub.status.idle":"2024-12-06T10:56:19.858196Z","shell.execute_reply.started":"2024-12-06T10:56:19.717870Z","shell.execute_reply":"2024-12-06T10:56:19.857379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the counts for each anatomy\nanatomy_counts = train_df['anatomy'].value_counts()\n\n# Define labels and sizes\nsize = anatomy_counts.values\nlabels = anatomy_counts.index\n\n# Plotting bar chart\nplt.figure(figsize=(7, 5))\nplt.bar(labels, size, color=[\"r\", \"c\"])\n\n# Title and labels\nplt.title(\"Anatomy Distribution\", fontsize=18)\nplt.xlabel(\"Anatomy\", fontsize=14)\nplt.ylabel(\"Count\", fontsize=14)\n\n# Rotate x-axis labels\nplt.xticks(rotation=45)  # You can change the angle to any value you prefer\n\n# Print counts\nfor anatomy, count in zip(labels, size):\n    print(f\"{anatomy.capitalize()} Cases = {count}\")\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:19.859212Z","iopub.execute_input":"2024-12-06T10:56:19.859558Z","iopub.status.idle":"2024-12-06T10:56:20.061939Z","shell.execute_reply.started":"2024-12-06T10:56:19.859521Z","shell.execute_reply":"2024-12-06T10:56:20.061060Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Preprocess .csv files 📐\r\n\r\n## 4.1 Add Image Path\r\nThis will help access the images in the feature.","metadata":{}},{"cell_type":"code","source":"# Step 1: Filter train.csv\nmalignant = train_df[train_df['target'] == 1]\nbenign = train_df[train_df['target'] == 0]\n\nrequired_benign = 15000 - len(malignant)\nbenign = benign.sample(n=required_benign, random_state=42)\n\ntrain_reduced = pd.concat([malignant, benign])\nprint(f\"Reduced Train Size: {len(train_reduced)}\")\n\n# Step 2: Filter test.csv\ntest_reduced = test_df.sample(n=8000, random_state=42)\nprint(f\"Reduced Test Size: {len(test_reduced)}\")\n\n# Step 3: Copy Required Images\ndef manage_images(src_folder, dst_folder, valid_image_names):\n    os.makedirs(dst_folder, exist_ok=True)\n    valid_image_paths = set(valid_image_names + \".jpg\")  # Convert image names to full file names\n    for img_file in os.listdir(src_folder):\n        src_path = os.path.join(src_folder, img_file)\n        dst_path = os.path.join(dst_folder, img_file)\n        if img_file in valid_image_paths:\n            shutil.copy(src_path, dst_path)\n\n# Copy train images\nmanage_images(\n    src_folder=\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\",\n    dst_folder=\"/kaggle/working/train\",\n    valid_image_names=train_reduced['dcm_name'].values  # Pass the column as a list\n)\n\n# Copy test images\nmanage_images(\n    src_folder=\"/kaggle/input/siim-isic-melanoma-classification/jpeg/test\",\n    dst_folder=\"/kaggle/working/test\",\n    valid_image_names=test_reduced['dcm_name'].values  # Pass the column as a list\n)\n\n# Save reduced CSV files\ntrain_reduced.to_csv(\"/kaggle/working/train_reduced.csv\", index=False)\ntest_reduced.to_csv(\"/kaggle/working/test_reduced.csv\", index=False)\n\nprint(\"Filtered data and images have been saved successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:31:55.315113Z","iopub.execute_input":"2024-12-06T16:31:55.315442Z","iopub.status.idle":"2024-12-06T16:38:47.305017Z","shell.execute_reply.started":"2024-12-06T16:31:55.315411Z","shell.execute_reply":"2024-12-06T16:38:47.304109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify the number of images in the reduced train and test datasets\ntrain_images = os.listdir(\"/kaggle/working/train\")\ntest_images = os.listdir(\"/kaggle/working/test\")\n\n# Print counts\nprint(f\"Number of train images after reduction: {len(train_images)}\")\nprint(f\"Number of test images after reduction: {len(test_images)}\")\n\n# Display a few train image samples\nprint(\"\\nSample train images:\")\nprint(train_images[:5])\n\n# Display a few test image samples\nprint(\"\\nSample test images:\")\nprint(test_images[:5])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:40:23.994235Z","iopub.execute_input":"2024-12-06T16:40:23.994581Z","iopub.status.idle":"2024-12-06T16:40:24.012622Z","shell.execute_reply.started":"2024-12-06T16:40:23.994548Z","shell.execute_reply":"2024-12-06T16:40:24.011938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # # === DICOM ===\n# # # Create the paths\n# # path_train = directory + '/train/' + train_df['dcm_name'] + '.dcm'\n# # path_test = directory + '/test/' + test_df['dcm_name'] + '.dcm'\n\n# # Append to the original dataframes\n# # train_df['path_dicom'] = path_train\n# # test_df['path_dicom'] = path_test\n\n# # === JPEG ===\n# # Create the paths\n# path_train = directory + '/jpeg/train/' + train_df['dcm_name'] + '.jpg'\n# path_test = directory + '/jpeg/test/' + test_df['dcm_name'] + '.jpg'\n\n# # # Append to the original dataframes\n# train_df['path_jpeg'] = path_train\n# test_df['path_jpeg'] = path_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:20.229270Z","iopub.status.idle":"2024-12-06T10:56:20.229554Z","shell.execute_reply.started":"2024-12-06T10:56:20.229417Z","shell.execute_reply":"2024-12-06T10:56:20.229431Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a new folder for reduced images\nreduced_image_folder = \"/path/to/reduced_images\"\nos.makedirs(reduced_image_folder, exist_ok=True)\n\n# Assuming `train_reduced` and `test_reduced` are the reduced dataframes with the 'image_name' column\n\n# Path to the original train image folder\noriginal_train_folder = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\"\noriginal_test_folder = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/test\"\n\n# Function to copy images to the reduced image folder and update paths in the dataframe\ndef update_image_paths_and_copy_images(data, original_folder, reduced_folder):\n    updated_paths = []\n    \n    # Loop through each image in the reduced dataset\n    for image_name in data['dcm_name']:\n        original_image_path = os.path.join(original_folder, f\"{image_name}.jpg\")\n        new_image_path = os.path.join(reduced_folder, f\"{image_name}.jpg\")\n        \n        # Copy the image to the new folder\n        shutil.copy(original_image_path, new_image_path)\n        \n        # Update the path in the list\n        updated_paths.append(new_image_path)\n    \n    # Add the updated paths to the dataframe\n    data['image_path'] = updated_paths\n    return data\n\n# Update train and test data with new image paths\ntrain_reduced_updated = update_image_paths_and_copy_images(train_reduced, original_train_folder, reduced_image_folder)\ntest_reduced_updated = update_image_paths_and_copy_images(test_reduced, original_test_folder, reduced_image_folder)\n\n# Save the updated data to new CSV files with updated image paths\ntrain_reduced_updated.to_csv(\"train_reduced_updated.csv\", index=False)\ntest_reduced_updated.to_csv(\"test_reduced_updated.csv\", index=False)\n\nprint(\"Updated CSV files saved as train_reduced_updated.csv and test_reduced_updated.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:40:36.969233Z","iopub.execute_input":"2024-12-06T16:40:36.969925Z","iopub.status.idle":"2024-12-06T16:45:49.758506Z","shell.execute_reply.started":"2024-12-06T16:40:36.969891Z","shell.execute_reply":"2024-12-06T16:45:49.757636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create new folders for reduced DICOM files\nreduced_dicom_train_folder = \"/path/to/reduced_dicom_train\"\nreduced_dicom_test_folder = \"/path/to/reduced_dicom_test\"\n\n# Create the folders if they do not exist\nos.makedirs(reduced_dicom_train_folder, exist_ok=True)\nos.makedirs(reduced_dicom_test_folder, exist_ok=True)\n\n# Assuming `train_reduced` and `test_reduced` are the reduced dataframes with the 'dcm_name' column\n\n# Path to the original train/test DICOM folder\noriginal_dicom_train_folder = \"/kaggle/input/siim-isic-melanoma-classification/train\"\noriginal_dicom_test_folder = '/kaggle/input/siim-isic-melanoma-classification/test'\n\n# Function to copy DICOM files to the reduced folder and update paths in the dataframe\ndef update_dicom_paths_and_copy_files(data, original_dicom_folder, reduced_dicom_folder):\n    updated_dicom_paths = []\n    \n    # Loop through each DICOM file in the reduced dataset\n    for dcm_name in data['dcm_name']:\n        original_dcm_path = os.path.join(original_dicom_folder, f\"{dcm_name}.dcm\")\n        new_dcm_path = os.path.join(reduced_dicom_folder, f\"{dcm_name}.dcm\")\n        \n        # Copy the DICOM file to the new folder\n        shutil.copy(original_dcm_path, new_dcm_path)\n        \n        # Update the path in the list\n        updated_dicom_paths.append(new_dcm_path)\n    \n    # Add the updated DICOM paths to the dataframe\n    data['dcm_path'] = updated_dicom_paths\n    return data\n\n# Update train and test data with new DICOM paths\ntrain_reduced_updated_dcm = update_dicom_paths_and_copy_files(train_reduced, original_dicom_train_folder, reduced_dicom_train_folder)\ntest_reduced_updated_dcm = update_dicom_paths_and_copy_files(test_reduced, original_dicom_test_folder, reduced_dicom_test_folder)\n\n# Save the updated data to new CSV files with updated DICOM paths\ntrain_reduced_updated_dcm.to_csv(\"train_reduced_updated_dcm.csv\", index=False)\ntest_reduced_updated_dcm.to_csv(\"test_reduced_updated_dcm.csv\", index=False)\n\nprint(\"Updated CSV files with DICOM paths saved as 'train_reduced_updated_dcm.csv' and 'test_reduced_updated_dcm.csv'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:45:49.760457Z","iopub.execute_input":"2024-12-06T16:45:49.761017Z","iopub.status.idle":"2024-12-06T16:58:48.506638Z","shell.execute_reply.started":"2024-12-06T16:45:49.760974Z","shell.execute_reply":"2024-12-06T16:58:48.505739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train  = train_reduced_updated \ndf_test  = test_reduced_updated \n\ndf_train.to_csv(\"df_train.csv\", index=False)\ndf_test.to_csv(\"df_train.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:58:48.507722Z","iopub.execute_input":"2024-12-06T16:58:48.508008Z","iopub.status.idle":"2024-12-06T16:58:48.624426Z","shell.execute_reply.started":"2024-12-06T16:58:48.507980Z","shell.execute_reply":"2024-12-06T16:58:48.623494Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4.2 One Hot Encoding\r\nTransforming all categorical features un numerical.\r\n> Note1: `sex`, `anatomy`, `diagnosis` need to be encoded.\r\n\r\n> Note2: `benign_malignant` column will be dropped, as the information is already in the `target` column.","metadata":{}},{"cell_type":"markdown","source":"# Train ","metadata":{}},{"cell_type":"code","source":"to_encode = ['sex', 'anatomy', 'diagnosis']\nencoded_all = []\n\nlabel_encoder = LabelEncoder()\n\nfor column in to_encode:\n    encoded = label_encoder.fit_transform(df_train[column])\n    encoded_all.append(encoded)\n    \ndf_train['sex'] = encoded_all[0]\ndf_train['anatomy'] = encoded_all[1]\ndf_train['diagnosis'] = encoded_all[2]\n\nif 'benign_malignant' in train_reduced.columns : train_reduced.drop(['benign_malignant'], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:58:48.626251Z","iopub.execute_input":"2024-12-06T16:58:48.626522Z","iopub.status.idle":"2024-12-06T16:58:48.651444Z","shell.execute_reply.started":"2024-12-06T16:58:48.626496Z","shell.execute_reply":"2024-12-06T16:58:48.650737Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# test","metadata":{}},{"cell_type":"code","source":"to_encode = ['sex', 'anatomy']\nencoded_all = []\n\nlabel_encoder = LabelEncoder()\n\nfor column in to_encode:\n    encoded = label_encoder.fit_transform(df_test[column])\n    encoded_all.append(encoded)\n    \ndf_test['sex'] = encoded_all[0]\ndf_test['anatomy'] = encoded_all[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:58:48.652489Z","iopub.execute_input":"2024-12-06T16:58:48.652800Z","iopub.status.idle":"2024-12-06T16:58:48.662311Z","shell.execute_reply.started":"2024-12-06T16:58:48.652739Z","shell.execute_reply":"2024-12-06T16:58:48.661258Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" Save the files before continuing.","metadata":{}},{"cell_type":"code","source":"# # Save the files\n# train_df.to_csv('train_clean.csv', index=False)\n# test_reduced.to_csv('test_clean.csv', index=False)\n\n# # Display the contents of the train_df DataFrame\n# # train_df\n# # ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:20.233082Z","iopub.status.idle":"2024-12-06T10:56:20.233407Z","shell.execute_reply.started":"2024-12-06T10:56:20.233236Z","shell.execute_reply":"2024-12-06T10:56:20.233259Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_clean = train_df  # Define train_clean if you need it\n# train_clean.to_csv('train_clean.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:20.234264Z","iopub.status.idle":"2024-12-06T10:56:20.234606Z","shell.execute_reply.started":"2024-12-06T10:56:20.234437Z","shell.execute_reply":"2024-12-06T10:56:20.234453Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n\n# # Assuming data_dir is already defined\n# train_reduced['image_path'] = train_reduced['dcm_name'].apply(lambda x: os.path.join(directory, 'jpeg/train', f'{x}.jpg'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:20.237859Z","iopub.status.idle":"2024-12-06T10:56:20.238178Z","shell.execute_reply.started":"2024-12-06T10:56:20.238021Z","shell.execute_reply":"2024-12-06T10:56:20.238036Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. The Images 📸\r\n\r\nThere are 2 types of images containing the same information:\r\n1. `.dcm` files: [DICOM files](https://en.wikipedia.org/wiki/DICOM). It's saved in the \"Digital Imaging and Communications in Medicine\" format. It contains an image from a medical scan, such as an ultrasound or MRI + information about the patient.\r\n2. `.jpeg` files: the DICOM files converted into .jpeg format\r\n3. `.tfrec` files: [The TFRecord file format is a simple record-oriented binary format for ML training data.](https://docs.databricks.com/applications/deep-learning/data-prep/tfrecords-to-tensorflow.html#:~:text=The%20TFRecord%20file%20format%20is,part%20of%20an%20input%20pipeline.)\r\n\r\n## 5.1 Sanity Check\r\n> Check if images in `.dcm` and `.jpeg` format have the same number of observations as in `train_df` and `test_df`.","metadata":{}},{"cell_type":"code","source":"# print('Train .dcm number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/train'))), '\\n' +\n#       'Test .dcm number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/test'))), '\\n' +\n#       'Train .jpeg number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/jpeg/train'))), '\\n' +\n#       'Test .jpeg number of images:', len(list(os.listdir('../input/siim-isic-melanoma-classification/jpeg/test'))), '\\n' +\n#       '-----------------------', '\\n' +\n#       'There is the same number of images as in train/ test .csv datasets')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:20.239541Z","iopub.status.idle":"2024-12-06T10:56:20.239869Z","shell.execute_reply.started":"2024-12-06T10:56:20.239725Z","shell.execute_reply":"2024-12-06T10:56:20.239741Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5.2 DICOM Images\r\n\r\n### Malignant vs Benign Images\r\n\r\nLet's look at the difference between *malignant* and *benign* melanomas.","metadata":{}},{"cell_type":"code","source":"def show_images(data, n=5, rows=1, cols=5, title='Default'):\n    plt.figure(figsize=(16, 4))\n\n    for k, path in enumerate(df_train['dcm_path'][:n]):\n        image = pydicom.dcmread(path)  # Use dcmread to read the DICOM file\n        image = image.pixel_array  # Extract pixel data\n        \n        # Optionally resize the image if needed\n        # image = resize(image, (200, 200), anti_aliasing=True)\n\n        plt.suptitle(title, fontsize=16)\n        plt.subplot(rows, cols, k + 1)\n        plt.imshow(image, cmap='gray')  # Use gray colormap for medical images\n        plt.axis('off')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:58:48.679490Z","iopub.execute_input":"2024-12-06T16:58:48.679735Z","iopub.status.idle":"2024-12-06T16:58:48.688590Z","shell.execute_reply.started":"2024-12-06T16:58:48.679699Z","shell.execute_reply":"2024-12-06T16:58:48.687967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show Benign Samples\nshow_images(df_train[df_train['target'] == 0], n=10, rows=2, cols=5, title='Benign Sample')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:58:48.689623Z","iopub.execute_input":"2024-12-06T16:58:48.689941Z","iopub.status.idle":"2024-12-06T16:59:10.557410Z","shell.execute_reply.started":"2024-12-06T16:58:48.689913Z","shell.execute_reply":"2024-12-06T16:59:10.556514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show Malignant Samples\nshow_images(df_train[df_train['target'] == 1], n=10, rows=2, cols=5, title='Malignant Sample')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:59:10.558488Z","iopub.execute_input":"2024-12-06T16:59:10.558820Z","iopub.status.idle":"2024-12-06T16:59:31.563795Z","shell.execute_reply.started":"2024-12-06T16:59:10.558773Z","shell.execute_reply":"2024-12-06T16:59:31.562995Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Class Imbalance ⚖\r\n\r\nThis is a **very** important topic in this classification problem, as the 2 classes we are dealing with are highly imbalanced, with 98% of the data being *benign* and only 2% of the data being *malignant*.\r\n\r\n<img src='https://i.imgur.com/Oc4Z3EP.png' width=400>\r\n\r\nThis is also the kind of problem where you **DON'T** want to have False Negatives. It's waaayyy worse to tell a patient they don't have cancer when they actually do, than to tell em they do have it and they actually don't. So, having balanced classes is *crucial*.\r\n\r\n### We can do:\r\n* **Oversampling**: of the minority class, increasing the number of images through augmentations\r\n* **Understampling**: of the majority class (we shall see how the process is going)\r\n\r\n> <img src='https://i.imgur.com/OwvqMbQ.png' width=450>\r\n\r\n### What is Data Augmentation?\r\n\r\nIs *moving, rotating, cropping, flipping, changing color/brightness/hue and whatever else you can come up with* to change the aspect of the original image. It is helpful in Overfitting, as the model learns not only 1 aspect of the image, but multiple (a cat can be standing up straight, or funny upside down, in a b&w image etc.).\r\n> A child recognizes a cat in all these random contexts ... but *your* model can? 😉\r\n\r\n<img src=\"https://miro.medium.com/max/850/1*ae1tW5ngf1zhPRyh7aaM1Q.png\" width = 450>\r\n\r\n### Other things to keep in mind:\r\n<div class=\"alert alert-block alert-info\">\r\n<p><b>#1:</b> Different skin tones. Might need to find something that levels that.</p>\r\n<p><b>#2:</b> Different lightings in the image.</p>\r\n<p><b>#3:</b> Different sizes of the images. We need todiv>\r\n\r\n## 6.1 B&W View 🤍🖤","metadata":{}},{"cell_type":"code","source":"# class_weights = compute_class_weight('balanced', classes=np.unique(y_train), y=y_train)\n# class_weights = dict(enumerate(class_weights))\n# #","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:20.248071Z","iopub.status.idle":"2024-12-06T10:56:20.248381Z","shell.execute_reply.started":"2024-12-06T10:56:20.248209Z","shell.execute_reply":"2024-12-06T10:56:20.248246Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X_train_img, X_val_img, X_train_meta, X_val_meta, y_train, y_val = train_test_split(\n#     X_train_img, metadata, y_train, test_size=0.2, random_state=42\n# )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:20.249242Z","iopub.status.idle":"2024-12-06T10:56:20.249584Z","shell.execute_reply.started":"2024-12-06T10:56:20.249442Z","shell.execute_reply":"2024-12-06T10:56:20.249457Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Pre-trained EfficientNetB0\n# base_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(128, 128, 3))\n# base_model.trainable = False  # Freeze base model\n\n# # Image input\n# image_input = base_model.input\n# x = base_model(image_input, training=False)\n# x = GlobalAveragePooling2D()(x)  # Pooling\n\n# # Metadata input\n# meta_input = Input(shape=(X_train_meta.shape[1],))\n# y = Dense(32, activation='relu')(meta_input)\n\n# # Combine features\n# combined = concatenate([x, y])\n# z = Dense(64, activation='relu')(combined)\n# z = Dropout(0.5)(z)\n# output = Dense(1, activation='sigmoid')(z)  # Binary classification\n\n# # Create model\n# model = Model(inputs=[image_input, meta_input], outputs=output)\n# model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:20.250867Z","iopub.status.idle":"2024-12-06T10:56:20.251303Z","shell.execute_reply.started":"2024-12-06T10:56:20.251036Z","shell.execute_reply":"2024-12-06T10:56:20.251051Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Preprocess images and extract the target values\n# def preprocess_images(image_paths, target_size=(128, 128)):\n#     images = []\n#     for path in image_paths:\n#         img = load_img(path, target_size=target_size)  # Resize image to target size\n#         img = img_to_array(img) / 255.0  # Normalize the image\n#         images.append(img)\n#     return np.array(images)","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Extract image paths and preprocess them\n# X = preprocess_images(df_train['image_path'].values)\n# y = df_train['target'].values  # Use target column for labels","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Split the data into training and validation sets\n# X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Load pretrained model (e.g., VGG16)\n# base_model = VGG16(weights='imagenet', include_top=False, input_shape=(128, 128, 3))\n\n# # Freeze the base model layers\n# base_model.trainable = False\n","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Build the model\n# model = Sequential([\n#     base_model,\n#     GlobalAveragePooling2D(),  # Global pooling layer\n#     Dropout(0.2),  # Dropout to prevent overfitting\n#     Dense(128, activation='relu'),  # Fully connected layer\n#     Dense(1, activation='sigmoid')  # Output layer (binary classification)\n# ])","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Compile the model\n# model.compile(optimizer=Adam(learning_rate=0.0001), loss='binary_crossentropy', metrics=['accuracy'])\n\n# # Data augmentation setup\n# train_datagen = ImageDataGenerator(\n#     rescale=1./255,\n#     rotation_range=20,\n#     width_shift_range=0.1,\n#     height_shift_range=0.1,\n#     shear_range=0.1,\n#     zoom_range=0.1,\n#     fill_mode='nearest'\n# )","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create data generators for training and validation\n# train_generator = train_datagen.flow(X_train, y_train, batch_size=32)\n# val_generator = ImageDataGenerator(rescale=1./255).flow(X_val, y_val, batch_size=32)","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Train the model\n# history = model.fit(\n#     train_generator,\n#     validation_data=val_generator,\n#     epochs=10,\n#     steps_per_epoch=len(X_train) // 32,\n#     validation_steps=len(X_val) // 32\n# )\n\n# # Save the model\n# model.save('melanoma_classifier.h5')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:54:42.844079Z","iopub.execute_input":"2024-12-06T12:54:42.844451Z","iopub.status.idle":"2024-12-06T13:20:43.986740Z","shell.execute_reply.started":"2024-12-06T12:54:42.844417Z","shell.execute_reply":"2024-12-06T13:20:43.985954Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split the data","metadata":{}},{"cell_type":"code","source":"# Split the data into training and validation (80% train, 20% validation)\ntrain_df, val_df = train_test_split(df_train, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:59:31.574213Z","iopub.execute_input":"2024-12-06T16:59:31.574467Z","iopub.status.idle":"2024-12-06T16:59:31.592268Z","shell.execute_reply.started":"2024-12-06T16:59:31.574441Z","shell.execute_reply":"2024-12-06T16:59:31.591485Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# training and validation images","metadata":{}},{"cell_type":"code","source":"# Preprocess the images (resize and normalize)\ndef preprocess_images(image_paths, target_size=(224, 224)):\n    images = []\n    for path in image_paths:\n        img = load_img(path, target_size=target_size)\n        img_array = img_to_array(img)\n        img_array = img_array / 255.0  # Normalize pixel values\n        images.append(img_array)\n    return np.array(images)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:03:42.821072Z","iopub.execute_input":"2024-12-06T17:03:42.821377Z","iopub.status.idle":"2024-12-06T17:03:42.826382Z","shell.execute_reply.started":"2024-12-06T17:03:42.821352Z","shell.execute_reply":"2024-12-06T17:03:42.825459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess the training and validation images\nX_train = preprocess_images(train_df['image_path'].values)\nX_val = preprocess_images(val_df['image_path'].values)\n\n# Get the labels\ny_train = train_df['target'].values\ny_val = val_df['target'].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:03:44.889010Z","iopub.execute_input":"2024-12-06T17:03:44.889701Z","iopub.status.idle":"2024-12-06T17:25:28.913626Z","shell.execute_reply.started":"2024-12-06T17:03:44.889669Z","shell.execute_reply":"2024-12-06T17:25:28.912895Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EfficientNetB0 model ","metadata":{}},{"cell_type":"code","source":"# Build the model using EfficientNetB0\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\nbase_model.trainable = False  # Freeze the base model layers\n\n# Create the model on top of EfficientNetB0\nmodel = models.Sequential([\n    base_model,\n    layers.GlobalAveragePooling2D(),\n    layers.Dense(1024, activation='relu'),\n    layers.Dense(1, activation='sigmoid')  # Sigmoid for binary classification\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:30:28.562028Z","iopub.execute_input":"2024-12-06T17:30:28.562693Z","iopub.status.idle":"2024-12-06T17:30:29.694207Z","shell.execute_reply.started":"2024-12-06T17:30:28.562657Z","shell.execute_reply":"2024-12-06T17:30:29.693518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer=Adam(learning_rate=0.0001), loss='binary_crossentropy', metrics=['accuracy'])\n\n# Print the model summary\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:31:54.393337Z","iopub.execute_input":"2024-12-06T17:31:54.394164Z","iopub.status.idle":"2024-12-06T17:31:54.418126Z","shell.execute_reply.started":"2024-12-06T17:31:54.394130Z","shell.execute_reply":"2024-12-06T17:31:54.417422Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ImageDataGenerator","metadata":{}},{"cell_type":"code","source":"# Set up ImageDataGenerator for data augmentation on training data\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Set up validation data (no augmentation, only rescaling)\nval_datagen = ImageDataGenerator(rescale=1./255)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:33:53.585803Z","iopub.execute_input":"2024-12-06T17:33:53.586136Z","iopub.status.idle":"2024-12-06T17:33:53.590807Z","shell.execute_reply.started":"2024-12-06T17:33:53.586106Z","shell.execute_reply":"2024-12-06T17:33:53.589992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fit the model using the training and validation data\nhistory = model.fit(\n    train_datagen.flow(X_train, y_train, batch_size=32),\n    epochs=10,\n    validation_data=val_datagen.flow(X_val, y_val, batch_size=32)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:33:58.095695Z","iopub.execute_input":"2024-12-06T17:33:58.096492Z","iopub.status.idle":"2024-12-06T17:52:48.455886Z","shell.execute_reply.started":"2024-12-06T17:33:58.096457Z","shell.execute_reply":"2024-12-06T17:52:48.455158Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluate the model","metadata":{}},{"cell_type":"code","source":"# Evaluate the model on the validation set\nval_loss, val_acc = model.evaluate(val_datagen.flow(X_val, y_val, batch_size=32))\nprint(f\"Validation Loss: {val_loss}\")\nprint(f\"Validation Accuracy: {val_acc}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T17:52:48.457396Z","iopub.execute_input":"2024-12-06T17:52:48.457671Z","iopub.status.idle":"2024-12-06T17:52:51.354996Z","shell.execute_reply.started":"2024-12-06T17:52:48.457643Z","shell.execute_reply":"2024-12-06T17:52:51.354134Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  Save the model\n","metadata":{}},{"cell_type":"code","source":"# Save the model\nmodel.save('skin_cancer_model.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T18:13:47.409547Z","iopub.execute_input":"2024-12-06T18:13:47.409915Z","iopub.status.idle":"2024-12-06T18:13:47.744495Z","shell.execute_reply.started":"2024-12-06T18:13:47.409881Z","shell.execute_reply":"2024-12-06T18:13:47.743807Z"}},"outputs":[],"execution_count":null}]}