{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np \nimport cv2\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\nfrom tensorflow.keras.layers import Dense, Flatten, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom sklearn.utils import class_weight\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:31:56.422577Z","iopub.execute_input":"2023-04-09T23:31:56.422932Z","iopub.status.idle":"2023-04-09T23:32:02.090343Z","shell.execute_reply.started":"2023-04-09T23:31:56.422901Z","shell.execute_reply":"2023-04-09T23:32:02.08937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta/images paths\ndef map_imgs(meta_file_path, image_file_path):\n    IMAGE_PATH = image_file_path\n#     meta_file= '/kaggle/input/siim-isic-melanoma-classification/train.csv'\n#     IMAGE_PATH = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train'\n    \n    df = pd.read_csv(meta_file_path)\n\n    # adding new column = filepath (complete image path)\n    df['image_path'] = df['image_name'].map(lambda x:  os.path.join(IMAGE_PATH,x+'.jpg'))\n\n    # mapping dictionary\n    labelmap = {}\n    for l in df.target.unique().tolist():\n        if l == 1:\n            labelmap[l] = 'melanoma'\n        else:\n            labelmap[l] = 'benign'\n\n    # seperate list of image that are labelled = melanoma\n    df_melanoma = df[df['target'] == 1]\n    df_melanoma.reset_index(drop=True, inplace=True)\n\n    # view \n    print(f\"Found {df.groupby('target').count()['image_name'][1]} images that are labelled = melanoma and {df.groupby('target').count()['image_name'][0]} that are labelled = benign\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:32:06.79667Z","iopub.execute_input":"2023-04-09T23:32:06.797528Z","iopub.status.idle":"2023-04-09T23:32:06.817156Z","shell.execute_reply.started":"2023-04-09T23:32:06.797479Z","shell.execute_reply":"2023-04-09T23:32:06.815384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = map_imgs(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\",\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\")\ndf_train= df_train[[\"target\",\"image_path\"]]\nprint(df_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:32:09.245192Z","iopub.execute_input":"2023-04-09T23:32:09.245627Z","iopub.status.idle":"2023-04-09T23:32:09.472274Z","shell.execute_reply.started":"2023-04-09T23:32:09.245588Z","shell.execute_reply":"2023-04-09T23:32:09.471337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"target\"] = df_train[\"target\"].astype(str)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:32:12.673865Z","iopub.execute_input":"2023-04-09T23:32:12.674833Z","iopub.status.idle":"2023-04-09T23:32:12.708488Z","shell.execute_reply.started":"2023-04-09T23:32:12.674796Z","shell.execute_reply":"2023-04-09T23:32:12.707322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = df_train[df_train['target'] == \"1\"]\nss.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:32:15.108762Z","iopub.execute_input":"2023-04-09T23:32:15.109135Z","iopub.status.idle":"2023-04-09T23:32:15.121771Z","shell.execute_reply.started":"2023-04-09T23:32:15.109101Z","shell.execute_reply":"2023-04-09T23:32:15.120782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dividir el dataframe en dos según la categoría\nmelanoma = df_train[df_train['target'] == \"1\"]\nbenigno = df_train[df_train['target'] == \"0\"]\nprint(melanoma.shape[0])\n# Hacer una muestra aleatoria del 80% de cada categoría\nmelanoma_train = melanoma.sample(frac=0.8, random_state=1)\nprint(melanoma_train.shape[0])\nbenigno_train = benigno.sample(frac=0.8, random_state=1)\nprint(benigno_train.shape[0])\n#test_benigno = benigno_sub[~benigno_sub.index.isin(train_benigno.index)]\n\n# Unir las muestras de ambas categorías\ndf_train = pd.concat([melanoma_train, benigno_train])\nprint(df_train.shape[0])\n","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:32:17.950119Z","iopub.execute_input":"2023-04-09T23:32:17.950798Z","iopub.status.idle":"2023-04-09T23:32:17.972772Z","shell.execute_reply.started":"2023-04-09T23:32:17.950762Z","shell.execute_reply":"2023-04-09T23:32:17.971906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_generator = ImageDataGenerator(preprocessing_function=preprocess_input, validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:32:23.254912Z","iopub.execute_input":"2023-04-09T23:32:23.255307Z","iopub.status.idle":"2023-04-09T23:32:23.260326Z","shell.execute_reply.started":"2023-04-09T23:32:23.255277Z","shell.execute_reply":"2023-04-09T23:32:23.259302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = data_generator.flow_from_dataframe(df_train,\n                                              target_size= (224, 224),\n                                                     x_col='image_path',\n                                                     y_col='target',\n                                              batch_size = 128,\n                                              subset = 'training',\n                                              shuffle = True,\n                                              class_mode ='binary')\nval_generator = data_generator.flow_from_dataframe(df_train,\n                                              target_size= (224, 224),x_col='image_path',\n                                                     y_col='target',\n                                              batch_size = 128,\n                                              subset = 'validation',\n                                              shuffle = False,\n                                              class_mode ='binary')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:32:26.254417Z","iopub.execute_input":"2023-04-09T23:32:26.255252Z","iopub.status.idle":"2023-04-09T23:32:47.247148Z","shell.execute_reply.started":"2023-04-09T23:32:26.255202Z","shell.execute_reply":"2023-04-09T23:32:47.245508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_weights = class_weight.compute_class_weight('balanced',\n                                                 classes= np.unique(df_train[\"target\"]),\n                                                 y =df_train[\"target\"])","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:33:12.615333Z","iopub.execute_input":"2023-04-09T23:33:12.615792Z","iopub.status.idle":"2023-04-09T23:33:12.69683Z","shell.execute_reply.started":"2023-04-09T23:33:12.615753Z","shell.execute_reply":"2023-04-09T23:33:12.695867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert numpy array to dictionary\nclass_weight = dict(enumerate(class_weights.flatten(), 0))\nclass_weight","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:33:25.799404Z","iopub.execute_input":"2023-04-09T23:33:25.799756Z","iopub.status.idle":"2023-04-09T23:33:25.807699Z","shell.execute_reply.started":"2023-04-09T23:33:25.799726Z","shell.execute_reply":"2023-04-09T23:33:25.80666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = ResNet50(include_top=False, pooling='avg', weights='imagenet', input_shape=(224,224,3))\nfor layer in base_model.layers[:-4]:\n    layer.trainable = False\n# base_model.summary() ","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:33:42.973177Z","iopub.execute_input":"2023-04-09T23:33:42.973546Z","iopub.status.idle":"2023-04-09T23:33:47.657032Z","shell.execute_reply.started":"2023-04-09T23:33:42.973514Z","shell.execute_reply":"2023-04-09T23:33:47.656034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEPS_PER_EPOCH = 21201 // 128\nVALID_STEPS = 5300 // 128","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:34:05.233109Z","iopub.execute_input":"2023-04-09T23:34:05.233469Z","iopub.status.idle":"2023-04-09T23:34:05.238157Z","shell.execute_reply.started":"2023-04-09T23:34:05.23344Z","shell.execute_reply":"2023-04-09T23:34:05.236871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_model = Sequential([base_model])\n# my_model.add(GlobalAveragePooling2D())\nmy_model.add(Dense(512, activation='relu'))\nmy_model.add(Dropout(0.5))\nmy_model.add(Dense(1, activation='sigmoid'))\n\nmy_model.compile(optimizer=Adam(learning_rate=0.0001),\n             loss='binary_crossentropy',\n             metrics=['accuracy',tf.keras.metrics.Recall(),tf.keras.metrics.AUC()])\n\nmodel_name = \"melanoma_model.h5\"\ncheckpoint = ModelCheckpoint(model_name,\n                            monitor=\"val_loss\",\n                            mode=\"min\",\n                            save_best_only = True,\n                            verbose=1)\n\nearlystopping = EarlyStopping(monitor='val_loss',min_delta = 0, patience = 5, verbose = 1, restore_best_weights=True)\n\ntry:\n    history = my_model.fit(train_generator,\n                           epochs=5,\n                           steps_per_epoch=STEPS_PER_EPOCH,\n                           validation_data=val_generator,\n                           validation_steps=VALID_STEPS,\n                           callbacks=[checkpoint,earlystopping],\n                           class_weight=class_weight)\nexcept KeyboardInterrupt:\n    print(\"\\nTraining Stopped\")","metadata":{"execution":{"iopub.status.busy":"2023-04-09T23:34:14.201276Z","iopub.execute_input":"2023-04-09T23:34:14.201653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.image as mpimp\nimport matplotlib.pyplot as plt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,7))\nplt.semilogx( history.history['accuracy'])\nplt.xlabel('Learning rate')\nplt.ylabel('Loss')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['recall'])\nplt.title('Recall durante el entrenamiento')\nplt.xlabel('época')\nplt.ylabel('recall')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = mpimp.imread('/kaggle/input/siim-isic-melanoma-classification/jpeg/test/ISIC_0052060.jpg')\nplt.imshow(test)\nplt.axis(False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_prep_image(filename, img_shape=224):\n  img = tf.io.read_file(filename)\n  img = tf.image.decode_image(img)\n  img = tf.image.resize(img, size= [img_shape, img_shape])\n  img = img/255.\n  return img","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediccion = load_prep_image('/kaggle/input/siim-isic-melanoma-classification/jpeg/test/ISIC_0052060.jpg')\nexpanded_steak = tf.expand_dims(steak, axis=0)\npred = my_model.predict(expanded_steak)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_name = ['benigno','maligno'] \npred_class = image_name[int(tf.round(pred))]\npred_class","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport random\n#funcion para plotear imagenes random\ndef view_random_image(target_dir, target_class):\n  target_folder = target_dir + target_class\n  random_image = random.sample(os.listdir(target_folder), 1)\n  print(random_image)\n\n  img = mpimg.imread(target_folder + '/' + random_image[0])\n  plt.imshow(img)\n  plt.title(target_class)\n  plt.axis('off')\n  print(f\"Image shape: {img.shape}\")\n  return img","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}