{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":63056,"databundleVersionId":9094797,"sourceType":"competition"},{"sourceId":643971,"sourceType":"datasetVersion","datasetId":319080},{"sourceId":1193409,"sourceType":"datasetVersion","datasetId":679322},{"sourceId":2275763,"sourceType":"datasetVersion","datasetId":1370616},{"sourceId":7649273,"sourceType":"datasetVersion","datasetId":4459076}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.13"},"colab":{"provenance":[],"collapsed_sections":["ik6nWn9CsobA","gjumE5cDsobE"],"gpuType":"T4"},"accelerator":"GPU"},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport shutil\nfrom shutil import copyfile\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom PIL import Image","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","id":"5aOGEFygsoay","execution":{"iopub.status.busy":"2024-08-29T12:22:33.117830Z","iopub.execute_input":"2024-08-29T12:22:33.118302Z","iopub.status.idle":"2024-08-29T12:22:33.124782Z","shell.execute_reply.started":"2024-08-29T12:22:33.118266Z","shell.execute_reply":"2024-08-29T12:22:33.123647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we create the empty directories corresponding to each class label.  In case we run this notebook or function a few times, we'll delete the directory structure (and all files in it recursively) if it already exists.","metadata":{"id":"suQbshy-soa1"}},{"cell_type":"code","source":"!rm -rf /kaggle/working/*\n!rm -rf /kaggle/working/lesions/\n!rm -rf /kaggle/working/aug/\n\n!mkdir -p /kaggle/working/lesions/\n!mkdir -p /kaggle/working/lesions/benign/\n!mkdir -p /kaggle/working/lesions/malignant/","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:22:33.155019Z","iopub.execute_input":"2024-08-29T12:22:33.156252Z","iopub.status.idle":"2024-08-29T12:22:40.104844Z","shell.execute_reply.started":"2024-08-29T12:22:33.156201Z","shell.execute_reply":"2024-08-29T12:22:40.103189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /kaggle/working/aug/benign/\n!mkdir -p /kaggle/working/aug/malignant/","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:22:40.107409Z","iopub.execute_input":"2024-08-29T12:22:40.107806Z","iopub.status.idle":"2024-08-29T12:22:42.410783Z","shell.execute_reply.started":"2024-08-29T12:22:40.107772Z","shell.execute_reply":"2024-08-29T12:22:42.409094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define root directory\ndata_dir = '/kaggle/working/lesions'\n\n# Empty directory to prevent FileExistsError if the function is run several times\n#if os.path.exists(data_dir):\n#  shutil.rmtree(data_dir)\n\n# Create the empty dir for each skin lesion\n#for label in label_names:\n#    os.makedirs(os.path.join(data_dir, label)) # e.g. /kaggle/working/lesions/malignant","metadata":{"id":"Cb99YU5zsoa1","execution":{"iopub.status.busy":"2024-08-29T12:22:42.412724Z","iopub.execute_input":"2024-08-29T12:22:42.413138Z","iopub.status.idle":"2024-08-29T12:22:42.419571Z","shell.execute_reply.started":"2024-08-29T12:22:42.413102Z","shell.execute_reply":"2024-08-29T12:22:42.418356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_names = ['malignant', 'benign']\n\nlabel_names = sorted(label_names)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:22:42.422529Z","iopub.execute_input":"2024-08-29T12:22:42.422906Z","iopub.status.idle":"2024-08-29T12:22:42.432410Z","shell.execute_reply.started":"2024-08-29T12:22:42.422875Z","shell.execute_reply":"2024-08-29T12:22:42.431167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing the ISIC 2024 dataset","metadata":{}},{"cell_type":"code","source":"#all positives are kept\n#all biopsied (have iddx_2) negatives are kept\n\n#keep 15% of negatives without lesion_id assigned\nno_id_negative_keep_frac = 0.15\n\n#keep 30% of negatives with lesion_id\nlesion_id_negative_keep_frac = 0.3\n\n#upsample factor for positives\npositive_upsample_multiple = 22\n\n#slight positive scores for special negatives\nid_assigned_negative_score = 0.02\nbiopsied_negative_score = 0.1\nbiopsied_indeterminate_score = 0.2\n\ndef balance_train_set(df):\n    # Keep small part of Negatives with no lesion_id\n    df_target_0_no_id = df[(df['target'] == 0) & (df['lesion_id'].isna())].sample(frac=no_id_negative_keep_frac, random_state=42)\n    \n    # Keep a small part of Negatives with lesion_id, not biopsied\n    df_target_0_with_id = df[(df['target'] == 0) & (df['lesion_id'].notna()) & (df['iddx_2'].isna())].sample(frac=lesion_id_negative_keep_frac, random_state=42)\n    \n    df_target_0_biopsied_indt = df[(df['target'] == 0) & (df['iddx_2'].notna()) & (df['iddx_1'] != \"Indeterminate\")]\n    \n    df_target_0_biopsied_neg = df[(df['target'] == 0) & (df['iddx_2'].notna()) & (df['iddx_1'] == \"Indeterminate\")]\n\n    df_target_0_with_id = df_target_0_with_id.assign(target=id_assigned_negative_score)\n    df_target_0_biopsied_neg = df_target_0_biopsied_neg.assign(target=biopsied_negative_score)\n    df_target_0_biopsied_indt = df_target_0_biopsied_indt.assign(target=biopsied_indeterminate_score)\n    \n    # Keep all positives\n    df_target_1 = df[df['target'] == 1]\n    \n    # Add upsampling for positive cases\n    df_target_1_upsampled = pd.concat([df_target_1] * positive_upsample_multiple, ignore_index=True)\n    \n    df_target_0 = pd.concat([df_target_0_no_id, df_target_0_with_id, df_target_0_biopsied_neg, df_target_0_biopsied_indt])\n    df_target_0_sampled = df_target_0.sample(n=30000)\n\n    # Combine all subsets\n    return pd.concat([df_target_0_sampled, df_target_1_upsampled]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:22:42.433748Z","iopub.execute_input":"2024-08-29T12:22:42.434122Z","iopub.status.idle":"2024-08-29T12:22:42.448029Z","shell.execute_reply.started":"2024-08-29T12:22:42.434092Z","shell.execute_reply":"2024-08-29T12:22:42.446759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_isic=pd.read_csv(r'/kaggle/input/isic-2024-challenge/train-metadata.csv')\ndf_isic = balance_train_set(df_isic)\nprint(df_isic.head())\nprint(df_isic.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:22:42.449362Z","iopub.execute_input":"2024-08-29T12:22:42.449714Z","iopub.status.idle":"2024-08-29T12:22:51.897690Z","shell.execute_reply.started":"2024-08-29T12:22:42.449684Z","shell.execute_reply":"2024-08-29T12:22:51.896438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (df_isic.columns)\n# Add .jpg extension to the image filenames\ndf_isic['isic_id']=df_isic['isic_id'].apply(lambda x: x+ '.jpg')\nx = df_isic.head()\nprint (df_isic.columns[1:])","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:22:51.898974Z","iopub.execute_input":"2024-08-29T12:22:51.899327Z","iopub.status.idle":"2024-08-29T12:22:51.932362Z","shell.execute_reply.started":"2024-08-29T12:22:51.899287Z","shell.execute_reply":"2024-08-29T12:22:51.931093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate over dataframe and move images to correct folders\nbenign_counter = 0\nfor index, row in tqdm(df_isic.iterrows(), total=df_isic.shape[0], desc=f'Copying ISIC 2024 dataset images..'):\n    # Get the image pathname\n    hot_label_int = row['target']\n    image_name = row['isic_id']\n    \n    if (hot_label_int == 0):\n\n        #if (benign_counter > 30000):\n        #    continue\n\n        hot_label = 'benign'\n        src_path = os.path.join(\"/kaggle/input/isic-2024-challenge/train-image/image\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n                \n        benign_counter += 1\n        \n    elif (hot_label_int == 1):\n\n        hot_label = 'malignant'  \n        src_path = os.path.join(\"/kaggle/input/isic-2024-challenge/train-image/image\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n        \n    else:\n        hot_label = 'benign'\n        src_path = os.path.join(\"/kaggle/input/isic-2024-challenge/train-image/image\", image_name)\n        dst_path = os.path.join(data_dir, hot_label, image_name)\n        copyfile(src_path, dst_path)\n                \n        benign_counter += 1\n        ","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:22:51.934026Z","iopub.execute_input":"2024-08-29T12:22:51.934497Z","iopub.status.idle":"2024-08-29T12:26:44.386243Z","shell.execute_reply.started":"2024-08-29T12:22:51.934453Z","shell.execute_reply":"2024-08-29T12:26:44.384813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot = 0\nfor label in label_names:\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    print(f\"There are {cnt_label} images with label {label}.\")\n    tot += cnt_label\nprint(f\"\\nThere are {tot} total images across all labels.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:26:44.387763Z","iopub.execute_input":"2024-08-29T12:26:44.388152Z","iopub.status.idle":"2024-08-29T12:26:44.418424Z","shell.execute_reply.started":"2024-08-29T12:26:44.388122Z","shell.execute_reply":"2024-08-29T12:26:44.417293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 224","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:26:44.424050Z","iopub.execute_input":"2024-08-29T12:26:44.424434Z","iopub.status.idle":"2024-08-29T12:26:44.429699Z","shell.execute_reply.started":"2024-08-29T12:26:44.424401Z","shell.execute_reply":"2024-08-29T12:26:44.428407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:26:44.431802Z","iopub.execute_input":"2024-08-29T12:26:44.432285Z","iopub.status.idle":"2024-08-29T12:26:44.446571Z","shell.execute_reply.started":"2024-08-29T12:26:44.432232Z","shell.execute_reply":"2024-08-29T12:26:44.445214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from PIL import Image\nimport matplotlib.image\n\ndef make_dataframe(sdir):\n    \n    filepaths=[]\n    labels=[]\n    classlist=sorted(os.listdir(sdir) )     \n    for klass in classlist:\n        classpath=os.path.join(sdir, klass) \n        if os.path.isdir(classpath):\n            flist=sorted(os.listdir(classpath)) \n            desc=f'{klass:25s}'\n            for f in tqdm(flist, ncols=130,desc=desc, unit='files', colour='blue'):\n                fpath=os.path.join(classpath,f)\n                filepaths.append(fpath)\n                labels.append(klass)\n    Fseries=pd.Series(filepaths, name='filepath')\n    Lseries=pd.Series(labels, name='label')\n    df=pd.concat([Fseries, Lseries], axis=1) \n    \n    return df\n\ndef make_dataframe_2(sdir, df_in):\n    \n    filepaths=[]\n    labels=[]\n    filenames=[]\n    isic_ids = []\n    classlist=sorted(os.listdir(sdir) )     \n    for klass in classlist:\n        classpath=os.path.join(sdir, klass) \n        if os.path.isdir(classpath):\n            flist=sorted(os.listdir(classpath)) \n            desc=f'{klass:25s}'\n            for f in tqdm(flist, ncols=130,desc=desc, unit='files', colour='blue'):\n                fpath=os.path.join(classpath,f)\n                filename=os.path.basename(fpath)\n                filepaths.append(fpath)\n                labels.append(klass)\n                if ('-' in filename):\n                    filename=filename.split('-')[0] + '.jpg'\n                #print(filename)                    \n                filenames.append(filename)\n                isic_id=os.path.splitext(filename)[0]\n                isic_ids.append(isic_id)\n    \n    print(len(filenames))\n    patient_ids = []\n    for i in range(len(filenames)):\n        file_name = filenames[i]\n        patient_id = list(df_in.query(\"isic_id == @file_name\")['patient_id'])[0]\n        patient_ids.append(patient_id)\n    \n    patient_id_series = pd.Series(patient_ids, name='patient_id')\n    isic_id_series=pd.Series(isic_ids, name='isic_id')\n    label_series=pd.Series(labels, name='label')\n    \n    Fseries=pd.Series(filepaths, name='filepath')\n    \n    print(isic_id_series)\n    print(Fseries)\n    print(patient_id_series)\n    print(label_series)\n    \n    df=pd.concat([isic_id_series, Fseries, patient_id_series, label_series], axis=1) \n    #df=pd.concat([isic_id_series, label_series], axis=1) \n    print(df.shape)\n    print(patient_id_series.shape)\n    \n    return df\n\n\ndef make_and_store_images(df, augdir, n,  img_size,  color_mode='rgb', save_prefix='aug-',save_format='jpg'):\n    df=df.copy()        \n    if os.path.isdir(augdir):# start with an empty directory\n        shutil.rmtree(augdir)\n    os.mkdir(augdir)  # if directory does not exist create it      \n    #for label in df['label'].unique():    \n    for label in label_names:\n        print(label)\n        classpath=os.path.join(augdir,label)    \n        os.mkdir(classpath) \n    total=0\n     \n    gen=ImageDataGenerator(horizontal_flip=True,  rotation_range=20, width_shift_range=.2,\n                                  height_shift_range=.2, zoom_range=.2)\n    groups=df.groupby('label')\n    for label in df['label'].unique():  \n        classdir=os.path.join(augdir, label)\n        group=groups.get_group(label)  # a dataframe holding only rows with the specified label \n        sample_count=len(group)   # determine how many samples there are in this class  \n            \n        if sample_count< n: # if the class has less than target number of images\n            aug_img_count=0\n            delta=n - sample_count  # number of augmented images to create            \n            msg='{0:40s} for class {1:^30s} creating {2:^5s} augmented images'.format(' ', label, str(delta))\n            print(msg, '\\r', end='') # prints over on the same line\n            aug_gen=gen.flow_from_dataframe( group,  x_col='filepath', y_col=None, target_size=img_size,class_mode=None, batch_size=1, shuffle=False, \n                                            save_to_dir=classdir, save_prefix=save_prefix, color_mode=color_mode,save_format=save_format)\n            while aug_img_count<delta:\n                images=next(aug_gen)            \n                aug_img_count += len(images)\n            total +=aug_img_count        \n    print('Total Augmented images created= ', total)\n    \ndef make_and_store_images_2(df, data_dir, augdir, n,  img_size,  color_mode='rgb', save_prefix='aug-',save_format='jpg'):\n    df=df.copy()        \n    if os.path.isdir(augdir):# start with an empty directory\n        shutil.rmtree(augdir)\n    os.mkdir(augdir)  # if directory does not exist create it      \n    #for label in df['label'].unique():    \n    for label in label_names:\n        print(label)\n        classpath=os.path.join(augdir,label)    \n        os.mkdir(classpath) \n    total=0\n     \n    gen=ImageDataGenerator(horizontal_flip=True, vertical_flip=True, rotation_range=20, width_shift_range=.2,\n                                  height_shift_range=.2, zoom_range=.2)\n    groups=df.groupby('label')\n    for label in df['label'].unique():\n        classdir_source = os.path.join(data_dir, label) \n        classdir=os.path.join(augdir, label)\n        group=groups.get_group(label)  # a dataframe holding only rows with the specified label \n        sample_count=len(group)   # determine how many samples there are in this class  \n        image_count = len(os.listdir(classdir_source)) \n        \n        if sample_count< n: # if the class has less than target number of images\n            aug_img_count=0\n            delta=n-sample_count  # number of augmented images to create            \n            msg='{0:40s} for class {1:^30s} creating {2:^5s} augmented images'.format(' ', label, str(delta))\n            print(msg, '\\r', end='') # prints over on the same line\n            \n            for index, row in tqdm(group.iterrows(), total=group.shape[0], desc=f'Augmenting images..'):\n                image_filename = row['filepath']\n                isic_id = row['isic_id']\n                image = Image.open(image_filename)\n                image = np.array(image)\n                #print(image.shape)\n                image = np.reshape(image, (1,) + image.shape)\n                #new_arr = np.expand_dims(my_arr, -1)\n                #print(image.shape)\n                \n                \n                aug_gen = gen.flow(image, y=None, batch_size=1, shuffle=False, sample_weight=None, seed=None,\n                            save_to_dir=classdir, save_prefix=image_filename, save_format=save_format, ignore_class_split=False, subset=None)\n                \n                n_iter = math.ceil(delta/sample_count)\n                \n                for i in range(n_iter):\n                    image_new = next(aug_gen)\n                    image_new_filename = classdir + '/' + isic_id + '-' + str(i) + '.jpg'\n                    #im = Image.fromarray(image_new)\n                    #im.save(image_new_filename)\n                    matplotlib.image.imsave(image_new_filename, image_new)\n                    #print(image_new_filename)\n                    total += 1\n                \n                #aug_gen=gen.flow_from_dataframe( group,  x_col='filepath', y_col=None, target_size=img_size,class_mode=None, batch_size=1, shuffle=False, \n                #                            save_to_dir=classdir, save_prefix=save_prefix, color_mode=color_mode,save_format=save_format)\n                #while aug_img_count<delta:\n                #    images=next(aug_gen)            \n                #    aug_img_count += len(images)\n                #total +=aug_img_count        \n    print('Total Augmented images created= ', total)\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:26:44.448651Z","iopub.execute_input":"2024-08-29T12:26:44.449084Z","iopub.status.idle":"2024-08-29T12:26:44.491181Z","shell.execute_reply.started":"2024-08-29T12:26:44.449037Z","shell.execute_reply":"2024-08-29T12:26:44.490076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"!rm -rf /kaggle/working/lesions/malignant/*-*.jpg*\n!rm -rf /kaggle/working/lesions/benign/*-*.jpg*","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:26:44.492820Z","iopub.execute_input":"2024-08-29T12:26:44.493326Z","iopub.status.idle":"2024-08-29T12:26:46.843469Z","shell.execute_reply.started":"2024-08-29T12:26:44.493263Z","shell.execute_reply":"2024-08-29T12:26:46.841999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sdir=data_dir\n#df=make_dataframe(sdir)\ndf_isic = df_isic.drop_duplicates(subset=[\"isic_id\"])\n\ndf=make_dataframe_2(sdir, df_isic)\nprint (df.head())\nprint ('length of dataframe is ',len(df))\nprint(df['label'].unique())","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:26:46.845354Z","iopub.execute_input":"2024-08-29T12:26:46.845815Z","iopub.status.idle":"2024-08-29T12:33:23.476736Z","shell.execute_reply.started":"2024-08-29T12:26:46.845769Z","shell.execute_reply":"2024-08-29T12:33:23.475597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/aug/malignant/*\n!rm -rf /kaggle/working/aug/benign/*","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:33:23.478054Z","iopub.execute_input":"2024-08-29T12:33:23.478402Z","iopub.status.idle":"2024-08-29T12:33:25.791567Z","shell.execute_reply.started":"2024-08-29T12:33:23.478373Z","shell.execute_reply":"2024-08-29T12:33:25.790070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"augdir=\"/kaggle/working/aug\" \nn=10000 \nimg_size=(IMG_SIZE,IMG_SIZE) \nmake_and_store_images_2(df, data_dir, augdir, n,  img_size,  color_mode='rgb', save_prefix='aug-',save_format='jpg')","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:33:25.793298Z","iopub.execute_input":"2024-08-29T12:33:25.793659Z","iopub.status.idle":"2024-08-29T12:34:33.238588Z","shell.execute_reply.started":"2024-08-29T12:33:25.793627Z","shell.execute_reply":"2024-08-29T12:34:33.237536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/aug/malignant/*.jpg_*\n!rm -rf /kaggle/working/aug/benign/*.jpg_*\n!rm -rf /kaggle/working/lesions/malignant/*.jpg_*\n!rm -rf /kaggle/working/lesions/benign/*.jpg_*\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:34:33.240147Z","iopub.execute_input":"2024-08-29T12:34:33.240489Z","iopub.status.idle":"2024-08-29T12:34:38.258181Z","shell.execute_reply.started":"2024-08-29T12:34:33.240459Z","shell.execute_reply":"2024-08-29T12:34:38.256558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug_path = \"/kaggle/working/aug/benign\"\nfiles_and_directories = os.listdir(aug_path)\nonly_files = [f for f in files_and_directories if os.path.isfile(os.path.join(aug_path, f))]\nfor image_name in tqdm(only_files, desc=f'Copying Augmented BNN images..'):\n    src_path = os.path.join(aug_path, image_name)\n    dst_path = os.path.join(data_dir, 'benign', image_name)\n    copyfile(src_path, dst_path)\n    \naug_path = \"/kaggle/working/aug/malignant\"\nfiles_and_directories = os.listdir(aug_path)\nonly_files = [f for f in files_and_directories if os.path.isfile(os.path.join(aug_path, f))]\nfor image_name in tqdm(only_files, desc=f'Copying Augmented MAL images..'):\n    src_path = os.path.join(aug_path, image_name)\n    dst_path = os.path.join(data_dir, 'malignant', image_name)\n    copyfile(src_path, dst_path)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:34:38.260179Z","iopub.execute_input":"2024-08-29T12:34:38.260628Z","iopub.status.idle":"2024-08-29T12:34:39.407043Z","shell.execute_reply.started":"2024-08-29T12:34:38.260591Z","shell.execute_reply":"2024-08-29T12:34:39.405742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_isic.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:34:39.408649Z","iopub.execute_input":"2024-08-29T12:34:39.409099Z","iopub.status.idle":"2024-08-29T12:34:39.415678Z","shell.execute_reply.started":"2024-08-29T12:34:39.409060Z","shell.execute_reply":"2024-08-29T12:34:39.414444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=make_dataframe_2(sdir, df_isic)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:34:39.417454Z","iopub.execute_input":"2024-08-29T12:34:39.417921Z","iopub.status.idle":"2024-08-29T12:43:20.062177Z","shell.execute_reply.started":"2024-08-29T12:34:39.417875Z","shell.execute_reply":"2024-08-29T12:43:20.061075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tot = 0\nfor label in label_names:\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    print(f\"There are {cnt_label} images with label {label}.\")\n    tot += cnt_label\nprint(f\"\\nThere are {tot} total images across all labels.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:43:20.063451Z","iopub.execute_input":"2024-08-29T12:43:20.063769Z","iopub.status.idle":"2024-08-29T12:43:20.098250Z","shell.execute_reply.started":"2024-08-29T12:43:20.063742Z","shell.execute_reply":"2024-08-29T12:43:20.097158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculating Class Weights","metadata":{}},{"cell_type":"code","source":"# Start weights at zero\nnum_classes = len(label_names)\nweights = [0] * num_classes  # e.g. [0, 0, 0, .. 0]\n\ntot = 0\nfor idx, label in enumerate(label_names):\n    cnt_label = len(os.listdir(os.path.join(data_dir, label)))\n    weights[idx] = cnt_label  # really a count right now\n    tot += cnt_label\n\nclass_frequencies = weights\nclass_frequencies = [ w / tot for w in weights ]  # [0.018897364771151177, 0.0297041 ...\n\nweights = [ 1.0 / cnt for cnt in weights ]\nweights = [ tot * w / num_classes for w in weights ]\n\nclass_weight = {}\nfor i in range(num_classes):\n    class_weight[i] = weights[i]\n    print(f\"Weight for class {i}: \" + '{:.2f}'.format(weights[i]))","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:43:20.099580Z","iopub.execute_input":"2024-08-29T12:43:20.099903Z","iopub.status.idle":"2024-08-29T12:43:20.137192Z","shell.execute_reply.started":"2024-08-29T12:43:20.099876Z","shell.execute_reply":"2024-08-29T12:43:20.136164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('benign_malignant.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:43:20.138652Z","iopub.execute_input":"2024-08-29T12:43:20.139089Z","iopub.status.idle":"2024-08-29T12:43:20.358924Z","shell.execute_reply.started":"2024-08-29T12:43:20.139058Z","shell.execute_reply":"2024-08-29T12:43:20.357720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/aug","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:43:20.360642Z","iopub.execute_input":"2024-08-29T12:43:20.361057Z","iopub.status.idle":"2024-08-29T12:43:21.810057Z","shell.execute_reply.started":"2024-08-29T12:43:20.361020Z","shell.execute_reply":"2024-08-29T12:43:21.808662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!rm -rf /kaggle/working/benign_vs_malignant.zip\n#!zip --quiet -r /kaggle/working/benign_vs_malignant.zip /kaggle/working/lesions/ ","metadata":{"execution":{"iopub.status.busy":"2024-08-29T12:43:21.812074Z","iopub.execute_input":"2024-08-29T12:43:21.812446Z","iopub.status.idle":"2024-08-29T12:43:21.817737Z","shell.execute_reply.started":"2024-08-29T12:43:21.812413Z","shell.execute_reply":"2024-08-29T12:43:21.816643Z"},"trusted":true},"execution_count":null,"outputs":[]}]}