{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np \nimport cv2\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\nfrom tensorflow.keras.layers import Dense, Flatten, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom sklearn.utils import class_weight\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-21T17:40:42.926135Z","iopub.execute_input":"2023-02-21T17:40:42.926503Z","iopub.status.idle":"2023-02-21T17:40:50.689987Z","shell.execute_reply.started":"2023-02-21T17:40:42.926423Z","shell.execute_reply":"2023-02-21T17:40:50.688906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta/images paths\ndef map_imgs(meta_file_path, image_file_path):\n    IMAGE_PATH = image_file_path\n#     meta_file= '/kaggle/input/siim-isic-melanoma-classification/train.csv'\n#     IMAGE_PATH = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train'\n    \n    df = pd.read_csv(meta_file_path)\n\n    # adding new column = filepath (complete image path)\n    df['image_path'] = df['image_name'].map(lambda x:  os.path.join(IMAGE_PATH,x+'.jpg'))\n\n    # mapping dictionary\n    labelmap = {}\n    for l in df.target.unique().tolist():\n        if l == 1:\n            labelmap[l] = 'melanoma'\n        else:\n            labelmap[l] = 'benign'\n\n    # seperate list of image that are labelled = melanoma\n    df_melanoma = df[df['target'] == 1]\n    df_melanoma.reset_index(drop=True, inplace=True)\n\n    # view \n    print(f\"Found {df.groupby('target').count()['image_name'][1]} images that are labelled = melanoma and {df.groupby('target').count()['image_name'][0]} that are labelled = benign\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:40:50.692134Z","iopub.execute_input":"2023-02-21T17:40:50.692821Z","iopub.status.idle":"2023-02-21T17:40:50.700553Z","shell.execute_reply.started":"2023-02-21T17:40:50.692790Z","shell.execute_reply":"2023-02-21T17:40:50.699377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = map_imgs(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\",\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\")\ndf_train= df_train[[\"target\",\"image_path\"]]\ndf_train[\"target\"] = df_train[\"target\"].astype(str)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:40:50.702189Z","iopub.execute_input":"2023-02-21T17:40:50.702878Z","iopub.status.idle":"2023-02-21T17:40:50.948297Z","shell.execute_reply.started":"2023-02-21T17:40:50.702828Z","shell.execute_reply":"2023-02-21T17:40:50.947322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = ImageDataGenerator(preprocessing_function=preprocess_input, validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:40:50.950943Z","iopub.execute_input":"2023-02-21T17:40:50.951815Z","iopub.status.idle":"2023-02-21T17:40:50.957409Z","shell.execute_reply.started":"2023-02-21T17:40:50.951785Z","shell.execute_reply":"2023-02-21T17:40:50.956260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = data_generator.flow_from_dataframe(df_train,\n                                              target_size= (224, 224),\n                                                     x_col='image_path',\n                                                     y_col='target',\n                                              batch_size = 128,\n                                              subset = 'training',\n                                              shuffle = True,\n                                              class_mode ='binary')\nval_generator = data_generator.flow_from_dataframe(df_train,\n                                              target_size= (224, 224),x_col='image_path',\n                                                     y_col='target',\n                                              batch_size = 128,\n                                              subset = 'validation',\n                                              shuffle = False,\n                                              class_mode ='binary')","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:40:50.958935Z","iopub.execute_input":"2023-02-21T17:40:50.959354Z","iopub.status.idle":"2023-02-21T17:42:39.899063Z","shell.execute_reply.started":"2023-02-21T17:40:50.959320Z","shell.execute_reply":"2023-02-21T17:42:39.897966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_weights = class_weight.compute_class_weight('balanced',\n                                                 classes= np.unique(df_train[\"target\"]),\n                                                 y =df_train[\"target\"])","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:42:39.901069Z","iopub.execute_input":"2023-02-21T17:42:39.901971Z","iopub.status.idle":"2023-02-21T17:42:39.954118Z","shell.execute_reply.started":"2023-02-21T17:42:39.901931Z","shell.execute_reply":"2023-02-21T17:42:39.953264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert numpy array to dictionary\nclass_weight = dict(enumerate(class_weights.flatten(), 0))\nclass_weight","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:42:39.957054Z","iopub.execute_input":"2023-02-21T17:42:39.957357Z","iopub.status.idle":"2023-02-21T17:42:39.964747Z","shell.execute_reply.started":"2023-02-21T17:42:39.957331Z","shell.execute_reply":"2023-02-21T17:42:39.963723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = ResNet50(include_top=False, pooling='avg', weights='imagenet', input_shape=(224,224,3))\nfor layer in base_model.layers[:-4]:\n    layer.trainable = False\n# base_model.summary() \n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:42:39.966237Z","iopub.execute_input":"2023-02-21T17:42:39.966890Z","iopub.status.idle":"2023-02-21T17:42:44.593153Z","shell.execute_reply.started":"2023-02-21T17:42:39.966849Z","shell.execute_reply":"2023-02-21T17:42:44.592078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEPS_PER_EPOCH = 26501 // 128\nVALID_STEPS = 6625 // 128","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:42:44.598120Z","iopub.execute_input":"2023-02-21T17:42:44.600885Z","iopub.status.idle":"2023-02-21T17:42:44.607612Z","shell.execute_reply.started":"2023-02-21T17:42:44.600843Z","shell.execute_reply":"2023-02-21T17:42:44.606302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_model = Sequential([base_model])\n# my_model.add(GlobalAveragePooling2D())\nmy_model.add(Dense(512, activation='relu'))\nmy_model.add(Dropout(0.5))\nmy_model.add(Dense(1, activation='sigmoid'))\n\nmy_model.compile(optimizer=Adam(learning_rate=0.0001),\n             loss='binary_crossentropy',\n             metrics=['accuracy',tf.keras.metrics.Recall()])\n\nmodel_name = \"melanoma_model.h5\"\ncheckpoint = ModelCheckpoint(model_name,\n                            monitor=\"val_loss\",\n                            mode=\"min\",\n                            save_best_only = True,\n                            verbose=1)\n\nearlystopping = EarlyStopping(monitor='val_loss',min_delta = 0, patience = 5, verbose = 1, restore_best_weights=True)\n\ntry:\n    history = my_model.fit(train_generator,\n                           epochs=5,\n                           steps_per_epoch=STEPS_PER_EPOCH,\n                           validation_data=val_generator,\n                           validation_steps=VALID_STEPS,\n                           callbacks=[checkpoint,earlystopping],\n                           class_weight=class_weight)\nexcept KeyboardInterrupt:\n    print(\"\\nTraining Stopped\")","metadata":{"execution":{"iopub.status.busy":"2023-02-21T17:42:44.614410Z","iopub.execute_input":"2023-02-21T17:42:44.615095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}