{"metadata":{"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":63056,"databundleVersionId":9094797,"sourceType":"competition"},{"sourceId":643971,"sourceType":"datasetVersion","datasetId":319080},{"sourceId":1193409,"sourceType":"datasetVersion","datasetId":679322},{"sourceId":2275763,"sourceType":"datasetVersion","datasetId":1370616},{"sourceId":7649273,"sourceType":"datasetVersion","datasetId":4459076}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.13"},"colab":{"provenance":[],"collapsed_sections":["ik6nWn9CsobA","gjumE5cDsobE"],"gpuType":"T4"},"accelerator":"GPU"},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport shutil\nfrom shutil import copyfile\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom PIL import Image\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","id":"5aOGEFygsoay","execution":{"iopub.status.busy":"2024-08-31T12:15:43.120520Z","iopub.execute_input":"2024-08-31T12:15:43.120798Z","iopub.status.idle":"2024-08-31T12:15:55.707784Z","shell.execute_reply.started":"2024-08-31T12:15:43.120773Z","shell.execute_reply":"2024-08-31T12:15:55.706753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing the HAM10000 dataset\n","metadata":{"id":"EdpIDyC8soay"}},{"cell_type":"code","source":"df=pd.read_csv(r'/kaggle/input/ham1000-segmentation-and-classification/GroundTruth.csv')\nprint (df.head())\nprint (len(df))\nprint (df.columns)\n\n \nlabel_names = ['malignant', 'benign']\n\nlabel_names = sorted(label_names)\n# print(label_names)\n\ndf['image']=df['image'].apply(lambda x: x+ '.jpg')\n# print (df.head())","metadata":{"id":"8dm8v9mMsoa0","outputId":"57b4998e-563c-469e-854b-5f0228eebfad","execution":{"iopub.status.busy":"2024-08-31T12:15:55.709439Z","iopub.execute_input":"2024-08-31T12:15:55.709959Z","iopub.status.idle":"2024-08-31T12:15:55.773556Z","shell.execute_reply.started":"2024-08-31T12:15:55.709931Z","shell.execute_reply":"2024-08-31T12:15:55.772643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we create the empty directories corresponding to each class label.  In case we run this notebook or function a few times, we'll delete the directory structure (and all files in it recursively) if it already exists.","metadata":{"id":"suQbshy-soa1"}},{"cell_type":"code","source":"!rm -rf /kaggle/working/*\n!rm -rf /kaggle/working/lesions/\n!rm -rf /kaggle/working/aug/\n\n!mkdir -p /kaggle/working/lesions/\n!mkdir -p /kaggle/working/lesions/benign/\n!mkdir -p /kaggle/working/lesions/malignant/","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:15:55.774718Z","iopub.execute_input":"2024-08-31T12:15:55.775073Z","iopub.status.idle":"2024-08-31T12:16:01.809019Z","shell.execute_reply.started":"2024-08-31T12:15:55.775041Z","shell.execute_reply":"2024-08-31T12:16:01.807871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /kaggle/working/aug/benign/\n!mkdir -p /kaggle/working/aug/malignant/","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:16:01.812606Z","iopub.execute_input":"2024-08-31T12:16:01.813191Z","iopub.status.idle":"2024-08-31T12:16:03.817772Z","shell.execute_reply.started":"2024-08-31T12:16:01.813159Z","shell.execute_reply":"2024-08-31T12:16:03.816549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define root directory\ndata_dir = '/kaggle/working/lesions'\n\n# Empty directory to prevent FileExistsError if the function is run several times\n#if os.path.exists(data_dir):\n#  shutil.rmtree(data_dir)\n\n# Create the empty dir for each skin lesion\n#for label in label_names:\n#    os.makedirs(os.path.join(data_dir, label)) # e.g. /kaggle/working/lesions/malignant","metadata":{"id":"Cb99YU5zsoa1","execution":{"iopub.status.busy":"2024-08-31T12:16:03.819247Z","iopub.execute_input":"2024-08-31T12:16:03.819565Z","iopub.status.idle":"2024-08-31T12:16:03.824273Z","shell.execute_reply.started":"2024-08-31T12:16:03.819535Z","shell.execute_reply":"2024-08-31T12:16:03.823316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_names = ['malignant', 'benign']\n\nlabel_names = sorted(label_names)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:16:03.825621Z","iopub.execute_input":"2024-08-31T12:16:03.826046Z","iopub.status.idle":"2024-08-31T12:16:03.838012Z","shell.execute_reply.started":"2024-08-31T12:16:03.826017Z","shell.execute_reply":"2024-08-31T12:16:03.837027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate over dataframe and move images to correct folders\nfor index, row in tqdm(df.iterrows(), total=df.shape[0], desc=f'Copying HAM10000 dataset images..'):\n    # Get the image pathname\n    hot_label = row[row == 1].index.tolist()[0]\n    image_name = row['image']\n    a = ['MEL', 'BCC', 'AKIEC', 'VASC']\n    if hot_label == 'NV':\n        pass\n    elif hot_label in a:\n        hot_label = 'malignant'\n        src_path = os.path.join(\"/kaggle/input/ham1000-segmentation-and-classification/images\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n    else:\n        hot_label = 'benign'\n        src_path = os.path.join(\"/kaggle/input/ham1000-segmentation-and-classification/images\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)","metadata":{"id":"EY_hfbuKsoa2","execution":{"iopub.status.busy":"2024-08-31T12:16:03.839278Z","iopub.execute_input":"2024-08-31T12:16:03.839797Z","iopub.status.idle":"2024-08-31T12:16:45.274773Z","shell.execute_reply.started":"2024-08-31T12:16:03.839767Z","shell.execute_reply":"2024-08-31T12:16:45.273877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot = 0\nfor label in label_names:\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    print(f\"There are {cnt_label} images with label {label}.\")\n    tot += cnt_label\nprint(f\"\\nThere are {tot} total images across all labels.\")","metadata":{"id":"NaTLW_3Zsoa3","outputId":"d9b039a2-6fcd-407f-d593-99615205302b","execution":{"iopub.status.busy":"2024-08-31T12:16:45.276007Z","iopub.execute_input":"2024-08-31T12:16:45.276303Z","iopub.status.idle":"2024-08-31T12:16:45.285011Z","shell.execute_reply.started":"2024-08-31T12:16:45.276278Z","shell.execute_reply":"2024-08-31T12:16:45.283980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing the ISIC 2019 dataset","metadata":{}},{"cell_type":"code","source":"df_isic=pd.read_csv(r'/kaggle/input/isic-2019/ISIC_2019_Training_GroundTruth.csv')\nprint (df_isic.columns)\n# Add .jpg extension to the image filenames\ndf_isic['image']=df_isic['image'].apply(lambda x: x+ '.jpg')\nx = df_isic.head()\nprint (df_isic.columns[1:])","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:16:45.286346Z","iopub.execute_input":"2024-08-31T12:16:45.286963Z","iopub.status.idle":"2024-08-31T12:16:45.354795Z","shell.execute_reply.started":"2024-08-31T12:16:45.286931Z","shell.execute_reply":"2024-08-31T12:16:45.353862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate over dataframe and move images to correct folders\nfor index, row in tqdm(df_isic.iterrows(), total=df_isic.shape[0], desc=f'Copying ISIC 2019 dataset images..'):\n    # Get the image pathname\n    hot_label = row[row == 1].index.tolist()[0]\n    image_name = row['image']\n    a = ['MEL', 'BCC', 'AKIEC', 'VASC','AK','SCC']\n    if hot_label == 'NV' or hot_label == 'UNK':\n        pass\n    elif hot_label in a:\n        hot_label = 'malignant'\n        src_path = os.path.join(\"/kaggle/input/isic-2019/ISIC_2019_Training_Input/ISIC_2019_Training_Input\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n    else:\n        hot_label = 'benign'\n        src_path = os.path.join(\"/kaggle/input/isic-2019/ISIC_2019_Training_Input/ISIC_2019_Training_Input\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:16:45.355952Z","iopub.execute_input":"2024-08-31T12:16:45.356255Z","iopub.status.idle":"2024-08-31T12:18:31.070305Z","shell.execute_reply.started":"2024-08-31T12:16:45.356229Z","shell.execute_reply":"2024-08-31T12:18:31.069426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot = 0\nfor label in label_names:\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    print(f\"There are {cnt_label} images with label {label}.\")\n    tot += cnt_label\nprint(f\"\\nThere are {tot} total images across all labels.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:18:31.074383Z","iopub.execute_input":"2024-08-31T12:18:31.074686Z","iopub.status.idle":"2024-08-31T12:18:31.088413Z","shell.execute_reply.started":"2024-08-31T12:18:31.074661Z","shell.execute_reply":"2024-08-31T12:18:31.087565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing the ISIC 2020 dataset","metadata":{}},{"cell_type":"code","source":"df_isic=pd.read_csv(r'/kaggle/input/siim-isic-melanoma-classification/train.csv')\nprint (df_isic.columns)\n# Add .jpg extension to the image filenames\ndf_isic['image_name']=df_isic['image_name'].apply(lambda x: x+ '.jpg')\nx = df_isic.head()\nprint (df_isic.columns[1:])","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:18:31.089998Z","iopub.execute_input":"2024-08-31T12:18:31.090301Z","iopub.status.idle":"2024-08-31T12:18:31.190832Z","shell.execute_reply.started":"2024-08-31T12:18:31.090277Z","shell.execute_reply":"2024-08-31T12:18:31.189933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_isic_malignant = df_isic.query('benign_malignant == \"malignant\"')\ndf_isic_benign = df_isic.query('benign_malignant == \"benign\"')\ndf_isic_benign = df_isic_benign.sample(n=10000)\ndf_isic = pd.concat([df_isic_benign, df_isic_malignant], axis = 0)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:18:31.192162Z","iopub.execute_input":"2024-08-31T12:18:31.192463Z","iopub.status.idle":"2024-08-31T12:18:31.229200Z","shell.execute_reply.started":"2024-08-31T12:18:31.192437Z","shell.execute_reply":"2024-08-31T12:18:31.228442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate over dataframe and move images to correct folders\nbenign_counter = 0\nfor index, row in tqdm(df_isic.iterrows(), total=df_isic.shape[0], desc=f'Copying ISIC 2020 dataset images..'):\n    # Get the image pathname\n    hot_label = row['benign_malignant']\n    image_name = row['image_name']\n    \n    if (hot_label == 'malignant'):\n        src_path = os.path.join(\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n        \n    elif (hot_label == 'benign'):\n        \n        #if (benign_counter > 5000):\n        #    continue\n            \n        src_path = os.path.join(\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n        benign_counter += 1\n        ","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:18:31.230271Z","iopub.execute_input":"2024-08-31T12:18:31.230558Z","iopub.status.idle":"2024-08-31T12:21:56.194512Z","shell.execute_reply.started":"2024-08-31T12:18:31.230535Z","shell.execute_reply":"2024-08-31T12:21:56.193581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot = 0\nfor label in label_names:\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    print(f\"There are {cnt_label} images with label {label}.\")\n    tot += cnt_label\nprint(f\"\\nThere are {tot} total images across all labels.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:21:56.195549Z","iopub.execute_input":"2024-08-31T12:21:56.195800Z","iopub.status.idle":"2024-08-31T12:21:56.213914Z","shell.execute_reply.started":"2024-08-31T12:21:56.195779Z","shell.execute_reply":"2024-08-31T12:21:56.213044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing the ISIC 2024 dataset","metadata":{}},{"cell_type":"code","source":"#all positives are kept\n#all biopsied (have iddx_2) negatives are kept\n\n#keep 15% of negatives without lesion_id assigned\nno_id_negative_keep_frac = 0.15\n\n#keep 30% of negatives with lesion_id\nlesion_id_negative_keep_frac = 0.3\n\n#upsample factor for positives\npositive_upsample_multiple = 22\n\n#slight positive scores for special negatives\nid_assigned_negative_score = 0.02\nbiopsied_negative_score = 0.1\nbiopsied_indeterminate_score = 0.2\n\ndef balance_train_set(df):\n    # Keep small part of Negatives with no lesion_id\n    df_target_0_no_id = df[(df['target'] == 0) & (df['lesion_id'].isna())].sample(frac=no_id_negative_keep_frac, random_state=42)\n    \n    # Keep a small part of Negatives with lesion_id, not biopsied\n    df_target_0_with_id = df[(df['target'] == 0) & (df['lesion_id'].notna()) & (df['iddx_2'].isna())].sample(frac=lesion_id_negative_keep_frac, random_state=42)\n    \n    df_target_0_biopsied_indt = df[(df['target'] == 0) & (df['iddx_2'].notna()) & (df['iddx_1'] != \"Indeterminate\")]\n    \n    df_target_0_biopsied_neg = df[(df['target'] == 0) & (df['iddx_2'].notna()) & (df['iddx_1'] == \"Indeterminate\")]\n\n    df_target_0_with_id = df_target_0_with_id.assign(target=id_assigned_negative_score)\n    df_target_0_biopsied_neg = df_target_0_biopsied_neg.assign(target=biopsied_negative_score)\n    df_target_0_biopsied_indt = df_target_0_biopsied_indt.assign(target=biopsied_indeterminate_score)\n    \n    # Keep all positives\n    df_target_1 = df[df['target'] == 1]\n    \n    # Add upsampling for positive cases\n    df_target_1_upsampled = pd.concat([df_target_1] * positive_upsample_multiple, ignore_index=True)\n    \n    df_target_0 = pd.concat([df_target_0_no_id, df_target_0_with_id, df_target_0_biopsied_neg, df_target_0_biopsied_indt])\n    df_target_0_sampled = df_target_0.sample(n=30000)\n\n    # Combine all subsets\n    return pd.concat([df_target_0_sampled, df_target_1_upsampled]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:21:56.215427Z","iopub.execute_input":"2024-08-31T12:21:56.215758Z","iopub.status.idle":"2024-08-31T12:21:56.226998Z","shell.execute_reply.started":"2024-08-31T12:21:56.215728Z","shell.execute_reply":"2024-08-31T12:21:56.226133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_isic=pd.read_csv(r'/kaggle/input/isic-2024-challenge/train-metadata.csv')\n\ndf_isic = balance_train_set(df_isic)\n\nprint(df_isic.head())\nprint(df_isic.shape)\n\nprint (df_isic.columns)\n# Add .jpg extension to the image filenames\ndf_isic['isic_id']=df_isic['isic_id'].apply(lambda x: x+ '.jpg')\nx = df_isic.head()\n#print (df_isic.columns[1:])","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:21:56.228116Z","iopub.execute_input":"2024-08-31T12:21:56.228366Z","iopub.status.idle":"2024-08-31T12:22:04.089643Z","shell.execute_reply.started":"2024-08-31T12:21:56.228339Z","shell.execute_reply":"2024-08-31T12:22:04.088659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate over dataframe and move images to correct folders\nbenign_counter = 0\nfor index, row in tqdm(df_isic.iterrows(), total=df_isic.shape[0], desc=f'Copying ISIC 2024 dataset images..'):\n    # Get the image pathname\n    hot_label_int = row['target']\n    image_name = row['isic_id']\n    \n    if (hot_label_int == 0):\n\n        #if (benign_counter > 5000):\n        #    continue\n\n        hot_label = 'benign'\n        src_path = os.path.join(\"/kaggle/input/isic-2024-challenge/train-image/image\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n                \n        benign_counter += 1\n        \n    elif (hot_label_int == 1):\n\n        hot_label = 'malignant'  \n        src_path = os.path.join(\"/kaggle/input/isic-2024-challenge/train-image/image\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n        \n","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:22:04.090648Z","iopub.execute_input":"2024-08-31T12:22:04.090927Z","iopub.status.idle":"2024-08-31T12:24:46.573323Z","shell.execute_reply.started":"2024-08-31T12:22:04.090905Z","shell.execute_reply":"2024-08-31T12:24:46.572333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot = 0\nfor label in label_names:\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    print(f\"There are {cnt_label} images with label {label}.\")\n    tot += cnt_label\nprint(f\"\\nThere are {tot} total images across all labels.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:24:46.574643Z","iopub.execute_input":"2024-08-31T12:24:46.574936Z","iopub.status.idle":"2024-08-31T12:24:46.610228Z","shell.execute_reply.started":"2024-08-31T12:24:46.574912Z","shell.execute_reply":"2024-08-31T12:24:46.609411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing the Melanoma-Cancer dataset","metadata":{"id":"5pBtZ8Eusoa5"}},{"cell_type":"code","source":"# Not using Benign images from this dataset due to the chance of making the final model biased\n# Iterate over the new dataset and copy only the Benign images\nbenign_paths = [\n                \"/kaggle/input/melanoma-cancer-dataset/test/Benign\",\n                \"/kaggle/input/melanoma-cancer-dataset/train/Benign\"\n                ]\n\nfor directory_path in benign_paths:\n\n    files_and_directories = os.listdir(directory_path)\n\n\n    only_files = [f for f in files_and_directories if os.path.isfile(os.path.join(directory_path, f))]\n    for image_name in tqdm(only_files, desc=f'Copying Melanoma-Cancer dataset benign images..'):\n\n        # Copy the image to the right label directory\n        src_path = os.path.join(directory_path, image_name)\n        dst_path = os.path.join(data_dir, 'benign', image_name)\n        copyfile(src_path, dst_path)","metadata":{"id":"9hocQnv0soa6","outputId":"c617c563-9a1c-43a6-e545-0b6ced861faa","execution":{"iopub.status.busy":"2024-08-31T12:24:46.611347Z","iopub.execute_input":"2024-08-31T12:24:46.611688Z","iopub.status.idle":"2024-08-31T12:25:12.435249Z","shell.execute_reply.started":"2024-08-31T12:24:46.611656Z","shell.execute_reply":"2024-08-31T12:25:12.434408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate over the new dataset and copy only the Malignant Images images\nmalignant_paths = [\n                  \"/kaggle/input/melanoma-cancer-dataset/test/Malignant\",\n                  \"/kaggle/input/melanoma-cancer-dataset/train/Malignant\"\n                   ]\n\nfor directory_path in malignant_paths:\n\n    files_and_directories = os.listdir(directory_path)\n\n\n    only_files = [f for f in files_and_directories if os.path.isfile(os.path.join(directory_path, f))]\n    for image_name in tqdm(only_files, desc=f'Copying Melanoma-Cancer dataset Malignant images..'):\n\n        # Copy the image to the right label directory\n        src_path = os.path.join(directory_path, image_name)\n        dst_path = os.path.join(data_dir, 'malignant', image_name)\n        copyfile(src_path, dst_path)","metadata":{"id":"T18gmPH_soa6","outputId":"7c7969f9-cfde-4d81-eb08-bbbd50ff6107","execution":{"iopub.status.busy":"2024-08-31T12:25:12.436500Z","iopub.execute_input":"2024-08-31T12:25:12.436783Z","iopub.status.idle":"2024-08-31T12:25:33.727546Z","shell.execute_reply.started":"2024-08-31T12:25:12.436759Z","shell.execute_reply":"2024-08-31T12:25:33.726646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot = 0\nfor label in label_names:\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    print(f\"There are {cnt_label} images with label {label}.\")\n    tot += cnt_label\nprint(f\"\\nThere are {tot} total images across all labels.\")","metadata":{"id":"h64nPY-3soa8","outputId":"76b9ceb5-44bc-46c0-fc94-b0eda3d1d877","execution":{"iopub.status.busy":"2024-08-31T12:25:33.728711Z","iopub.execute_input":"2024-08-31T12:25:33.729014Z","iopub.status.idle":"2024-08-31T12:25:33.770340Z","shell.execute_reply.started":"2024-08-31T12:25:33.728990Z","shell.execute_reply":"2024-08-31T12:25:33.769526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 224","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:33.771591Z","iopub.execute_input":"2024-08-31T12:25:33.772253Z","iopub.status.idle":"2024-08-31T12:25:33.776759Z","shell.execute_reply.started":"2024-08-31T12:25:33.772220Z","shell.execute_reply":"2024-08-31T12:25:33.775876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_dataframe(sdir):\n    \n    filepaths=[]\n    labels=[]\n    classlist=sorted(os.listdir(sdir) )     \n    for klass in classlist:\n        classpath=os.path.join(sdir, klass) \n        if os.path.isdir(classpath):\n            flist=sorted(os.listdir(classpath)) \n            desc=f'{klass:25s}'\n            for f in tqdm(flist, ncols=130,desc=desc, unit='files', colour='blue'):\n                fpath=os.path.join(classpath,f)\n                filepaths.append(fpath)\n                labels.append(klass)\n    Fseries=pd.Series(filepaths, name='filepaths')\n    Lseries=pd.Series(labels, name='labels')\n    df=pd.concat([Fseries, Lseries], axis=1) \n    \n    return df\n\ndef make_dataframe_2(sdir, df_in):\n    \n    filepaths=[]\n    labels=[]\n    filenames=[]\n    isic_ids = []\n    classlist=sorted(os.listdir(sdir) )     \n    for klass in classlist:\n        classpath=os.path.join(sdir, klass) \n        if os.path.isdir(classpath):\n            flist=sorted(os.listdir(classpath)) \n            desc=f'{klass:25s}'\n            for f in tqdm(flist, ncols=130,desc=desc, unit='files', colour='blue'):\n                fpath=os.path.join(classpath,f)\n                filename=os.path.basename(fpath)\n                filepaths.append(fpath)\n                labels.append(klass)\n                if ('-' in filename):\n                    filename=filename.split('-')[0] + '.jpg'\n                #print(filename)                    \n                filenames.append(filename)\n                isic_id=os.path.splitext(filename)[0]\n                isic_ids.append(isic_id)\n    \n    #print(len(filenames))\n    patient_ids = []\n    for i in range(len(filenames)):\n        file_name = filenames[i]\n        patient_id = list(df_in.query(\"isic_id == @file_name\")['patient_id'])[0]\n        patient_ids.append(patient_id)\n    \n    patient_id_series = pd.Series(patient_ids, name='patient_id')\n    isic_id_series=pd.Series(isic_ids, name='isic_id')\n    label_series=pd.Series(labels, name='label')\n    \n    Fseries=pd.Series(filepaths, name='filepath')\n    \n    #print(isic_id_series)\n    #print(Fseries)\n    #print(patient_id_series)\n    #print(label_series)\n    \n    df=pd.concat([isic_id_series, Fseries, patient_id_series, label_series], axis=1) \n    #df=pd.concat([isic_id_series, label_series], axis=1) \n    #print(df.shape)\n    #print(patient_id_series.shape)\n    \n    return df\n\ndef make_and_store_images(df, augdir, n,  img_size,  color_mode='rgb', save_prefix='aug-',save_format='jpg'):\n    df=df.copy()        \n    if os.path.isdir(augdir):# start with an empty directory\n        shutil.rmtree(augdir)\n    os.mkdir(augdir)  # if directory does not exist create it      \n    for label in df['labels'].unique():    \n        classpath=os.path.join(augdir,label)    \n        os.mkdir(classpath) \n    total=0\n     \n    gen=ImageDataGenerator(horizontal_flip=True,  rotation_range=20, width_shift_range=.2,\n                                  height_shift_range=.2, zoom_range=.2)\n    groups=df.groupby('labels')\n    for label in df['labels'].unique():  \n        classdir=os.path.join(augdir, label)\n        group=groups.get_group(label)  # a dataframe holding only rows with the specified label \n        sample_count=len(group)   # determine how many samples there are in this class  \n        if sample_count< n: # if the class has less than target number of images\n            aug_img_count=0\n            delta=n - sample_count  # number of augmented images to create            \n            msg='{0:40s} for class {1:^30s} creating {2:^5s} augmented images'.format(' ', label, str(delta))\n            print(msg, '\\r', end='') # prints over on the same line\n            aug_gen=gen.flow_from_dataframe( group,  x_col='filepaths', y_col=None, target_size=img_size,class_mode=None, batch_size=1, shuffle=False, \n                                            save_to_dir=classdir, save_prefix=save_prefix, color_mode=color_mode,save_format=save_format)\n            while aug_img_count<delta:\n                images=next(aug_gen)            \n                aug_img_count += len(images)\n            total +=aug_img_count        \n    print('Total Augmented images created= ', total)\n    \n\ndef make_and_store_images_2(df, data_dir, augdir, n,  img_size,  color_mode='rgb', save_prefix='aug-',save_format='jpg'):\n    df=df.copy()        \n    if os.path.isdir(augdir):# start with an empty directory\n        shutil.rmtree(augdir)\n    os.mkdir(augdir)  # if directory does not exist create it      \n    #for label in df['label'].unique():    \n    for label in label_names:\n        print(label)\n        classpath=os.path.join(augdir,label)    \n        os.mkdir(classpath) \n    total=0\n    \n    groups=df.groupby('label')\n    for label in df['label'].unique():\n        classdir_source = os.path.join(data_dir, label) \n        group=groups.get_group(label)  # a dataframe holding only rows with the specified label \n        \n        for index, row in tqdm(group.iterrows(), total=group.shape[0], desc=f'Applying CLAHE to images..'):\n            image_filename = row['filepath']\n            img = cv2.imread(image_filename, cv2.IMREAD_COLOR)\n            img = cv2.cvtColor(img, cv2.COLOR_RGB2Lab)\n\n            #configure CLAHE\n            clahe = cv2.createCLAHE(clipLimit=2.0,tileGridSize=(8,8))\n\n            #0 to 'L' channel, 1 to 'a' channel, and 2 to 'b' channel\n            img[:,:,0] = clahe.apply(img[:,:,0])\n\n            img = cv2.cvtColor(img, cv2.COLOR_Lab2RGB)\n\n            matplotlib.image.imsave(image_filename, img)\n                    \n    #data augmentation\n    gen=ImageDataGenerator(horizontal_flip=True, vertical_flip=True, rotation_range=20, width_shift_range=.2,\n                                  height_shift_range=.2, zoom_range=.2)\n    groups=df.groupby('label')\n    for label in df['label'].unique():\n        classdir_source = os.path.join(data_dir, label) \n        classdir=os.path.join(augdir, label)\n        group=groups.get_group(label)  # a dataframe holding only rows with the specified label \n        sample_count=len(group)   # determine how many samples there are in this class  \n        image_count = len(os.listdir(classdir_source)) \n        \n        if sample_count< n: # if the class has less than target number of images\n            aug_img_count=0\n            delta=n-sample_count  # number of augmented images to create            \n            msg='{0:40s} for class {1:^30s} creating {2:^5s} augmented images'.format(' ', label, str(delta))\n            print(msg, '\\r', end='') # prints over on the same line\n            \n            for index, row in tqdm(group.iterrows(), total=group.shape[0], desc=f'Augmenting images..'):\n                image_filename = row['filepath']\n                isic_id = row['isic_id']\n                image = Image.open(image_filename)\n                image = np.array(image)\n                #print(image.shape)\n                image = np.reshape(image, (1,) + image.shape)\n                #new_arr = np.expand_dims(my_arr, -1)\n                #print(image.shape)\n                \n                \n                aug_gen = gen.flow(image, y=None, batch_size=1, shuffle=False, sample_weight=None, seed=None,\n                            save_to_dir=classdir, save_prefix=image_filename, save_format=save_format, ignore_class_split=False, subset=None)\n                \n                n_iter = math.ceil(delta/sample_count)\n                \n                for i in range(n_iter):\n                    image_new = next(aug_gen)\n                    image_new_filename = classdir + '/' + isic_id + '-' + str(i) + '.jpg'\n                    #im = Image.fromarray(image_new)\n                    #im.save(image_new_filename)\n                    matplotlib.image.imsave(image_new_filename, image_new)\n                    #print(image_new_filename)\n                    total += 1\n                \n                #aug_gen=gen.flow_from_dataframe( group,  x_col='filepath', y_col=None, target_size=img_size,class_mode=None, batch_size=1, shuffle=False, \n                #                            save_to_dir=classdir, save_prefix=save_prefix, color_mode=color_mode,save_format=save_format)\n                #while aug_img_count<delta:\n                #    images=next(aug_gen)            \n                #    aug_img_count += len(images)\n                #total +=aug_img_count        \n    print('Total Augmented images created= ', total)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:33.777929Z","iopub.execute_input":"2024-08-31T12:25:33.778198Z","iopub.status.idle":"2024-08-31T12:25:33.812590Z","shell.execute_reply.started":"2024-08-31T12:25:33.778176Z","shell.execute_reply":"2024-08-31T12:25:33.811611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/lesions/malignant/*-*.jpg*\n!rm -rf /kaggle/working/lesions/benign/*-*.jpg*","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:33.813700Z","iopub.execute_input":"2024-08-31T12:25:33.813983Z","iopub.status.idle":"2024-08-31T12:25:35.869201Z","shell.execute_reply.started":"2024-08-31T12:25:33.813960Z","shell.execute_reply":"2024-08-31T12:25:35.867910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sdir=data_dir\ndf_isic = df_isic.drop_duplicates(subset=[\"isic_id\"])\n\ndf=make_dataframe(sdir)\nprint (df.head())\nprint ('length of dataframe is ',len(df))\n\naugdir=\"/kaggle/working/aug\" \nn=16000 \nimg_size=(IMG_SIZE,IMG_SIZE) \n#make_and_store_images(df, augdir, n,  img_size,  color_mode='rgb', save_prefix='aug-',save_format='jpg')","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:35.870752Z","iopub.execute_input":"2024-08-31T12:25:35.871077Z","iopub.status.idle":"2024-08-31T12:25:36.151704Z","shell.execute_reply.started":"2024-08-31T12:25:35.871048Z","shell.execute_reply":"2024-08-31T12:25:36.150873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/aug/malignant/*\n!rm -rf /kaggle/working/aug/benign/*","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:36.152873Z","iopub.execute_input":"2024-08-31T12:25:36.153197Z","iopub.status.idle":"2024-08-31T12:25:38.149989Z","shell.execute_reply.started":"2024-08-31T12:25:36.153173Z","shell.execute_reply":"2024-08-31T12:25:38.148763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"augdir=\"/kaggle/working/aug\" \nn=16000 \nimg_size=(IMG_SIZE,IMG_SIZE) \n#make_and_store_images(df, augdir, n,  img_size,  color_mode='rgb', save_prefix='aug-',save_format='jpg')","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:38.151676Z","iopub.execute_input":"2024-08-31T12:25:38.152214Z","iopub.status.idle":"2024-08-31T12:25:38.157633Z","shell.execute_reply.started":"2024-08-31T12:25:38.152172Z","shell.execute_reply":"2024-08-31T12:25:38.156685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/aug/malignant/*.jpg_*\n!rm -rf /kaggle/working/aug/benign/*.jpg_*\n!rm -rf /kaggle/working/lesions/malignant/*.jpg_*\n!rm -rf /kaggle/working/lesions/benign/*.jpg_*","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:38.158826Z","iopub.execute_input":"2024-08-31T12:25:38.159189Z","iopub.status.idle":"2024-08-31T12:25:42.246867Z","shell.execute_reply.started":"2024-08-31T12:25:38.159155Z","shell.execute_reply":"2024-08-31T12:25:42.245733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug_path = \"/kaggle/working/aug/benign\"\nfiles_and_directories = os.listdir(aug_path)\nonly_files = [f for f in files_and_directories if os.path.isfile(os.path.join(aug_path, f))]\nfor image_name in tqdm(only_files, desc=f'Copying Augmented BNN images..'):\n    src_path = os.path.join(aug_path, image_name)\n    dst_path = os.path.join(data_dir, 'benign', image_name)\n    copyfile(src_path, dst_path)\n    \naug_path = \"/kaggle/working/aug/malignant\"\nfiles_and_directories = os.listdir(aug_path)\nonly_files = [f for f in files_and_directories if os.path.isfile(os.path.join(aug_path, f))]\nfor image_name in tqdm(only_files, desc=f'Copying Augmented MAL images..'):\n    src_path = os.path.join(aug_path, image_name)\n    dst_path = os.path.join(data_dir, 'malignant', image_name)\n    copyfile(src_path, dst_path)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:42.248634Z","iopub.execute_input":"2024-08-31T12:25:42.249049Z","iopub.status.idle":"2024-08-31T12:25:42.265800Z","shell.execute_reply.started":"2024-08-31T12:25:42.249010Z","shell.execute_reply":"2024-08-31T12:25:42.264840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=make_dataframe(sdir)\nprint (df.head())\nprint ('length of dataframe is ',len(df))","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:42.266856Z","iopub.execute_input":"2024-08-31T12:25:42.267158Z","iopub.status.idle":"2024-08-31T12:25:42.532638Z","shell.execute_reply.started":"2024-08-31T12:25:42.267134Z","shell.execute_reply":"2024-08-31T12:25:42.531624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot = 0\nfor label in label_names:\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    print(f\"There are {cnt_label} images with label {label}.\")\n    tot += cnt_label\nprint(f\"\\nThere are {tot} total images across all labels.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:42.539021Z","iopub.execute_input":"2024-08-31T12:25:42.539289Z","iopub.status.idle":"2024-08-31T12:25:42.583466Z","shell.execute_reply.started":"2024-08-31T12:25:42.539266Z","shell.execute_reply":"2024-08-31T12:25:42.582591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculating Class Weights","metadata":{}},{"cell_type":"code","source":"# Start weights at zero\nnum_classes = len(label_names)\nweights = [0] * num_classes  # e.g. [0, 0, 0, .. 0]\n\ntot = 0\nfor idx, label in enumerate(label_names):\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    weights[idx] = cnt_label  # really a count right now\n    tot += cnt_label\n\nclass_frequencies = weights\nclass_frequencies = [ w / tot for w in weights ]  # [0.018897364771151177, 0.0297041 ...\n\nweights = [ 1.0 / cnt for cnt in weights ]\nweights = [ tot * w / num_classes for w in weights ]\n\nclass_weight = {}\nfor i in range(num_classes):\n    class_weight[i] = weights[i]\n    print(f\"Weight for class {i}: \" + '{:.2f}'.format(weights[i]))","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:42.584427Z","iopub.execute_input":"2024-08-31T12:25:42.584669Z","iopub.status.idle":"2024-08-31T12:25:42.628103Z","shell.execute_reply.started":"2024-08-31T12:25:42.584648Z","shell.execute_reply":"2024-08-31T12:25:42.627258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('benign_malignant.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:42.629244Z","iopub.execute_input":"2024-08-31T12:25:42.629566Z","iopub.status.idle":"2024-08-31T12:25:42.822127Z","shell.execute_reply.started":"2024-08-31T12:25:42.629535Z","shell.execute_reply":"2024-08-31T12:25:42.821200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!rm -rf /kaggle/working/benign_vs_malignant.zip\n#!zip --quiet -r /kaggle/working/benign_vs_malignant.zip /kaggle/working/lesions/ ","metadata":{"execution":{"iopub.status.busy":"2024-08-31T12:25:42.823274Z","iopub.execute_input":"2024-08-31T12:25:42.823535Z","iopub.status.idle":"2024-08-31T12:25:42.827461Z","shell.execute_reply.started":"2024-08-31T12:25:42.823513Z","shell.execute_reply":"2024-08-31T12:25:42.826418Z"},"trusted":true},"execution_count":null,"outputs":[]}]}