{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        path = os.path.join(dirname, filename)\nprint(path)\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-16T09:36:51.545994Z","iopub.execute_input":"2023-04-16T09:36:51.546293Z","iopub.status.idle":"2023-04-16T09:38:12.817924Z","shell.execute_reply.started":"2023-04-16T09:36:51.546264Z","shell.execute_reply":"2023-04-16T09:38:12.816577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np \nimport cv2\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\nfrom tensorflow.keras.layers import Dense, Flatten, GlobalAveragePooling2D, Dropout, Conv2D,MaxPooling2D\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom sklearn.utils import class_weight\nimport os\n\nfrom tensorflow.keras.applications import InceptionV3","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:38:48.831209Z","iopub.execute_input":"2023-04-16T09:38:48.831938Z","iopub.status.idle":"2023-04-16T09:38:57.638947Z","shell.execute_reply.started":"2023-04-16T09:38:48.831899Z","shell.execute_reply":"2023-04-16T09:38:57.637870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta/images paths\ndef map_imgs(meta_file_path, image_file_path):\n    IMAGE_PATH = image_file_path\n#     meta_file= '/kaggle/input/siim-isic-melanoma-classification/train.csv'\n#     IMAGE_PATH = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train'\n    \n    df = pd.read_csv(meta_file_path)\n\n    # adding new column = filepath (complete image path)\n    df['image_path'] = df['image_name'].map(lambda x:  os.path.join(IMAGE_PATH,x+'.jpg'))\n\n    # mapping dictionary\n    labelmap = {}\n    for l in df.target.unique().tolist():\n        if l == 1:\n            labelmap[l] = 'melanoma'\n        else:\n            labelmap[l] = 'benign'\n\n    # seperate list of image that are labelled = melanoma\n    df_melanoma = df[df['target'] == 1]\n    df_melanoma.reset_index(drop=True, inplace=True)\n\n    # view \n    print(f\"Found {df.groupby('target').count()['image_name'][1]} images that are labelled = melanoma and {df.groupby('target').count()['image_name'][0]} that are labelled = benign\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-16T07:05:32.711523Z","iopub.execute_input":"2023-04-16T07:05:32.711959Z","iopub.status.idle":"2023-04-16T07:05:32.720774Z","shell.execute_reply.started":"2023-04-16T07:05:32.711919Z","shell.execute_reply":"2023-04-16T07:05:32.719382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = map_imgs(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\",\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\")\ndf_train= df_train[[\"target\",\"image_path\"]]\ndf_train[\"target\"] = df_train[\"target\"].astype(str)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-04-16T07:18:45.167880Z","iopub.execute_input":"2023-04-16T07:18:45.168683Z","iopub.status.idle":"2023-04-16T07:18:45.378414Z","shell.execute_reply.started":"2023-04-16T07:18:45.168637Z","shell.execute_reply":"2023-04-16T07:18:45.377448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:39:02.668406Z","iopub.execute_input":"2023-04-16T09:39:02.669433Z","iopub.status.idle":"2023-04-16T09:39:02.835788Z","shell.execute_reply.started":"2023-04-16T09:39:02.669392Z","shell.execute_reply":"2023-04-16T09:39:02.834779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T07:20:06.966541Z","iopub.execute_input":"2023-04-16T07:20:06.966946Z","iopub.status.idle":"2023-04-16T07:20:06.983991Z","shell.execute_reply.started":"2023-04-16T07:20:06.966909Z","shell.execute_reply":"2023-04-16T07:20:06.982487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['image_path'] = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"+df_train['image_name']+\".jpg\"\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:39:07.128497Z","iopub.execute_input":"2023-04-16T09:39:07.129202Z","iopub.status.idle":"2023-04-16T09:39:07.167891Z","shell.execute_reply.started":"2023-04-16T09:39:07.129163Z","shell.execute_reply":"2023-04-16T09:39:07.166629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_input = df_train[['target','image_path']]\ndf_input[\"target\"] = df_input[\"target\"].astype(str)\ndf_input.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:39:09.998574Z","iopub.execute_input":"2023-04-16T09:39:09.999294Z","iopub.status.idle":"2023-04-16T09:39:10.039427Z","shell.execute_reply.started":"2023-04-16T09:39:09.999256Z","shell.execute_reply":"2023-04-16T09:39:10.038284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = ImageDataGenerator(preprocessing_function=preprocess_input, validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:39:13.168398Z","iopub.execute_input":"2023-04-16T09:39:13.169107Z","iopub.status.idle":"2023-04-16T09:39:13.174587Z","shell.execute_reply.started":"2023-04-16T09:39:13.169068Z","shell.execute_reply":"2023-04-16T09:39:13.173264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = data_generator.flow_from_dataframe(df_input,\n                                              target_size= (224, 224),\n                                                     x_col='image_path',\n                                                     y_col='target',\n                                              batch_size = 128,\n                                              subset = 'training',\n                                              shuffle = True,\n                                              class_mode ='binary')\nval_generator = data_generator.flow_from_dataframe(df_input,\n                                              target_size= (224, 224),\n                                                   x_col='image_path',\n                                                     y_col='target',\n                                              batch_size = 128,\n                                              subset = 'validation',\n                                              shuffle = False,\n                                              class_mode ='binary')","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:39:16.128475Z","iopub.execute_input":"2023-04-16T09:39:16.129194Z","iopub.status.idle":"2023-04-16T09:40:28.461875Z","shell.execute_reply.started":"2023-04-16T09:39:16.129153Z","shell.execute_reply":"2023-04-16T09:40:28.460831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_weights = class_weight.compute_class_weight('balanced',\n                                                 classes= np.unique(df_train[\"target\"]),\n                                                 y =df_train[\"target\"])","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:40:28.463818Z","iopub.execute_input":"2023-04-16T09:40:28.464457Z","iopub.status.idle":"2023-04-16T09:40:28.506907Z","shell.execute_reply.started":"2023-04-16T09:40:28.464412Z","shell.execute_reply":"2023-04-16T09:40:28.505904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert numpy array to dictionary\nclass_weight = dict(enumerate(class_weights.flatten(), 0))\nclass_weight","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:40:28.508293Z","iopub.execute_input":"2023-04-16T09:40:28.508749Z","iopub.status.idle":"2023-04-16T09:40:28.517686Z","shell.execute_reply.started":"2023-04-16T09:40:28.508707Z","shell.execute_reply":"2023-04-16T09:40:28.516568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device_name = tf.test.gpu_device_name()\nif len(device_name) > 0:\n    print(\"Found GPU at: {}\".format(device_name))\nelse:\n    device_name = \"/device:CPU:0\"\n    print(\"No GPU, using {}.\".format(device_name))","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:40:28.520417Z","iopub.execute_input":"2023-04-16T09:40:28.520928Z","iopub.status.idle":"2023-04-16T09:40:30.900954Z","shell.execute_reply.started":"2023-04-16T09:40:28.520892Z","shell.execute_reply":"2023-04-16T09:40:30.899407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_width = 224\nimg_height = 224\nwith tf.device(device_name):\n\n    base_model = InceptionV3(\n      weights='imagenet',\n      include_top=False,\n      input_shape=(img_width, img_height, 3))\n    for layer in base_model.layers:\n      layer.trainable = False\n\n    model = Sequential()\n    model.add(base_model)\n    model.add(GlobalAveragePooling2D())\n    model.add(Dense(128, activation='relu'))\n    model.add(Dropout(0.5))\n    model.add(Dense(1, activation='sigmoid'))        # Create a TensorFlow model.\n    \n    opt = Adam(learning_rate=0.0001)\n\n    model.compile(\n    optimizer=opt,\n    loss='binary_crossentropy',\n    metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:40:35.498020Z","iopub.execute_input":"2023-04-16T09:40:35.498394Z","iopub.status.idle":"2023-04-16T09:40:39.955372Z","shell.execute_reply.started":"2023-04-16T09:40:35.498361Z","shell.execute_reply":"2023-04-16T09:40:39.954286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GPU P100 Time","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom keras.callbacks import EarlyStopping\nearly_stop = EarlyStopping(patience=10, verbose=1)\n\nhistory = model.fit(\n    train_generator,\n    epochs=2,\n    validation_data=val_generator, \n    callbacks=[early_stop],\n    verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-16T09:40:41.678211Z","iopub.execute_input":"2023-04-16T09:40:41.678954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GPU TX2 Time","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom keras.callbacks import EarlyStopping\nearly_stop = EarlyStopping(patience=10, verbose=1)\n\nhistory = model.fit(\n    train_generator,\n    epochs=2,\n    validation_data=val_generator, \n    callbacks=[early_stop],\n    verbose=1)\n     ","metadata":{},"execution_count":null,"outputs":[]}]}