{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Modelo base","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pylab as plt\nimport os\nfrom os import listdir\nfrom os.path import isfile, join","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:36:35.805510Z","iopub.execute_input":"2021-11-24T22:36:35.806548Z","iopub.status.idle":"2021-11-24T22:36:35.812232Z","shell.execute_reply.started":"2021-11-24T22:36:35.806503Z","shell.execute_reply":"2021-11-24T22:36:35.811230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <font color=red>1. </font>Cargar las imágenes y los datos tabulares","metadata":{}},{"cell_type":"code","source":"# Resized images directory\ndir_2019_images = \"/kaggle/input/resizedsiimisic/train_resized/\"\nimages_2019 = [f for f in listdir(dir_2019_images) if isfile(join(dir_2019_images, f))]\n\n# CSV file\ntrain_df = pd.read_csv('/kaggle/input/resizedsiimisic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:36:35.814055Z","iopub.execute_input":"2021-11-24T22:36:35.814402Z","iopub.status.idle":"2021-11-24T22:36:43.334914Z","shell.execute_reply.started":"2021-11-24T22:36:35.814356Z","shell.execute_reply":"2021-11-24T22:36:43.334164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:36:43.336471Z","iopub.execute_input":"2021-11-24T22:36:43.336730Z","iopub.status.idle":"2021-11-24T22:36:43.356435Z","shell.execute_reply.started":"2021-11-24T22:36:43.336695Z","shell.execute_reply":"2021-11-24T22:36:43.355387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train shape:\", train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:36:43.359447Z","iopub.execute_input":"2021-11-24T22:36:43.360198Z","iopub.status.idle":"2021-11-24T22:36:43.364786Z","shell.execute_reply.started":"2021-11-24T22:36:43.360154Z","shell.execute_reply":"2021-11-24T22:36:43.364062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <font color=red>2. </font>Dividir conjunto de datos","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom sklearn.model_selection import train_test_split\n\nX = train_df\ny = train_df['target']","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:36:43.367972Z","iopub.execute_input":"2021-11-24T22:36:43.368448Z","iopub.status.idle":"2021-11-24T22:36:43.373797Z","shell.execute_reply.started":"2021-11-24T22:36:43.368413Z","shell.execute_reply":"2021-11-24T22:36:43.373035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split into train, validation and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, \n                                                    test_size=0.2, \n                                                    stratify=y,\n                                                    random_state=42)\n\nX_train, X_val, y_train, y_val = train_test_split(X_train, y_train, \n                                                  test_size=0.2,\n                                                  stratify=y_train,\n                                                  random_state=42)\n\nprint(\"Conjunto de train:\", X_train.shape)\nprint(\"Conjunto de validacion:\", X_val.shape)\nprint(\"Conjunto de prueba:\", X_test.shape)\nprint(\"-----------------------\")\nprint('Distribucion de train ->', Counter(y_train))\nprint('Distribucion de validacion ->', Counter(y_val))\nprint(\"Distribucion de prueba ->\", Counter(y_test))","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:36:43.375327Z","iopub.execute_input":"2021-11-24T22:36:43.375663Z","iopub.status.idle":"2021-11-24T22:36:43.426468Z","shell.execute_reply.started":"2021-11-24T22:36:43.375627Z","shell.execute_reply":"2021-11-24T22:36:43.425794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train[\"target\"] = X_train['target'].astype(str)\nX_val[\"target\"] = X_val['target'].astype(str)\nX_test[\"target\"] = X_test['target'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:36:43.427686Z","iopub.execute_input":"2021-11-24T22:36:43.428062Z","iopub.status.idle":"2021-11-24T22:36:43.461385Z","shell.execute_reply.started":"2021-11-24T22:36:43.428013Z","shell.execute_reply":"2021-11-24T22:36:43.460680Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <font color=red>3. </font>Crear y entrenar el modelo","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers.experimental import preprocessing\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import DenseNet121 # input size 224x224\nfrom tensorflow.keras.layers import Input, Dense, GlobalAveragePooling2D\nfrom tensorflow.keras.metrics import TruePositives, FalsePositives, TrueNegatives, FalseNegatives, AUC\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.losses import BinaryCrossentropy\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.models import Model","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:43:23.686857Z","iopub.execute_input":"2021-11-24T22:43:23.687168Z","iopub.status.idle":"2021-11-24T22:43:23.693692Z","shell.execute_reply.started":"2021-11-24T22:43:23.687130Z","shell.execute_reply":"2021-11-24T22:43:23.692608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(rescale=1./255.)\n\ntrain_generator = datagen.flow_from_dataframe(\n    dataframe=X_train,\n    directory=dir_2019_images,\n    x_col=\"image_name\",\n    y_col=\"target\",\n    batch_size=32,\n    seed=42,\n    shuffle=True,\n    class_mode=\"binary\"\n)\n\nvalid_generator = datagen.flow_from_dataframe(\n    dataframe=X_val,\n    directory=dir_2019_images,\n    x_col=\"image_name\",\n    y_col=\"target\",\n    batch_size=32,\n    seed=42,\n    shuffle=True,\n    class_mode=\"binary\"\n)\n\ntest_generator = datagen.flow_from_dataframe(\n    dataframe=X_test,\n    directory=dir_2019_images,\n    x_col=\"image_name\",\n    y_col=\"target\",\n    batch_size=32,\n    seed=42,\n    shuffle=False,\n    class_mode=\"binary\"\n)\n\nSTEP_SIZE_TRAIN = train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n//valid_generator.batch_size\nSTEP_SIZE_TEST = test_generator.n//test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:43:24.536817Z","iopub.execute_input":"2021-11-24T22:43:24.537629Z","iopub.status.idle":"2021-11-24T22:43:31.897882Z","shell.execute_reply.started":"2021-11-24T22:43:24.537583Z","shell.execute_reply":"2021-11-24T22:43:31.896413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = DenseNet121(input_shape=(None,None,3), \n                      include_top=False, \n                      weights='imagenet')","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:43:31.899709Z","iopub.execute_input":"2021-11-24T22:43:31.899967Z","iopub.status.idle":"2021-11-24T22:43:34.280005Z","shell.execute_reply.started":"2021-11-24T22:43:31.899931Z","shell.execute_reply":"2021-11-24T22:43:34.279112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for layer in encoder.layers:\n#    layer.trainable = False\n\ninputs = Input(shape=(None, None, 3))\nx = encoder(inputs, training=False)\nx = GlobalAveragePooling2D()(x)\npredictions = Dense(1, activation='sigmoid')(x)\nmodel = Model(inputs=inputs, outputs=predictions)","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:43:34.281780Z","iopub.execute_input":"2021-11-24T22:43:34.282056Z","iopub.status.idle":"2021-11-24T22:43:35.315441Z","shell.execute_reply.started":"2021-11-24T22:43:34.282022Z","shell.execute_reply":"2021-11-24T22:43:35.314711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:43:35.317368Z","iopub.execute_input":"2021-11-24T22:43:35.317612Z","iopub.status.idle":"2021-11-24T22:43:35.350481Z","shell.execute_reply.started":"2021-11-24T22:43:35.317581Z","shell.execute_reply":"2021-11-24T22:43:35.349793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"METRICS = [\n      TruePositives(name='tp'),\n      FalsePositives(name='fp'),\n      TrueNegatives(name='tn'),\n      FalseNegatives(name='fn'),\n      AUC(name='auc')\n]","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:43:35.351604Z","iopub.execute_input":"2021-11-24T22:43:35.351882Z","iopub.status.idle":"2021-11-24T22:43:35.374797Z","shell.execute_reply.started":"2021-11-24T22:43:35.351844Z","shell.execute_reply":"2021-11-24T22:43:35.374181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_filepath = '/kaggle/working/base_model.h5'\nmodel_checkpoint_callback = ModelCheckpoint(filepath=checkpoint_filepath,\n                                            monitor='val_auc',\n                                            mode='max',\n                                            verbose=1,\n                                            save_best_only=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:43:35.376049Z","iopub.execute_input":"2021-11-24T22:43:35.376316Z","iopub.status.idle":"2021-11-24T22:43:35.382584Z","shell.execute_reply.started":"2021-11-24T22:43:35.376284Z","shell.execute_reply":"2021-11-24T22:43:35.381872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer=Adam(),\n    loss=BinaryCrossentropy(),\n    metrics=METRICS\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:43:35.383951Z","iopub.execute_input":"2021-11-24T22:43:35.384265Z","iopub.status.idle":"2021-11-24T22:43:35.407786Z","shell.execute_reply.started":"2021-11-24T22:43:35.384226Z","shell.execute_reply":"2021-11-24T22:43:35.406928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_generator,  \n                    validation_data =valid_generator,\n                    steps_per_epoch=STEP_SIZE_TRAIN, \n                    validation_steps=STEP_SIZE_VALID,\n                    callbacks=[model_checkpoint_callback],\n                    epochs = 100)","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:46:13.651629Z","iopub.execute_input":"2021-11-24T22:46:13.651957Z","iopub.status.idle":"2021-11-24T22:47:40.241765Z","shell.execute_reply.started":"2021-11-24T22:46:13.651926Z","shell.execute_reply":"2021-11-24T22:47:40.240079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['auc'], \n         label='Training AUC (area = {:.3f})'.format(history.history['auc'][-1]))\nplt.plot(history.history['val_auc'], \n         label='Validation AUC ( area = {:.3f})'.format(history.history['val_auc'][-1]))\nplt.title('Base model')\nplt.ylabel('AUC')\nplt.xlabel('epoch')\nplt.grid()\nplt.legend(loc='best')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:47:40.242983Z","iopub.status.idle":"2021-11-24T22:47:40.243706Z","shell.execute_reply.started":"2021-11-24T22:47:40.243394Z","shell.execute_reply":"2021-11-24T22:47:40.243428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'], label='Training loss (loss = {:.3f})'.format(history.history['loss'][-1]))\nplt.plot(history.history['val_loss'], label='Validation loss (loss = {:.3f})'.format(history.history['val_loss'][-1]))\nplt.title('Model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.grid()\nplt.legend(loc='best')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:47:40.245228Z","iopub.status.idle":"2021-11-24T22:47:40.245898Z","shell.execute_reply.started":"2021-11-24T22:47:40.245653Z","shell.execute_reply":"2021-11-24T22:47:40.245677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <font color=red>4. </font>Evaluar el modelo","metadata":{}},{"cell_type":"code","source":"eval_metrics = model.evaluate(test_generator, \n                              steps=STEP_SIZE_TEST,\n                              return_dict=True,\n                              use_multiprocessing=False,\n                              verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:47:40.247096Z","iopub.status.idle":"2021-11-24T22:47:40.247585Z","shell.execute_reply.started":"2021-11-24T22:47:40.247351Z","shell.execute_reply":"2021-11-24T22:47:40.247374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Se obtiene las predicciones y las etiquetas del conjunto de prueba.","metadata":{}},{"cell_type":"code","source":"true_labels = test_generator.classes\npredict = model.predict(test_generator,\n                        verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:47:40.249202Z","iopub.status.idle":"2021-11-24T22:47:40.249621Z","shell.execute_reply.started":"2021-11-24T22:47:40.249389Z","shell.execute_reply":"2021-11-24T22:47:40.249419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Finalmente, se observa la curva ROC-AUC y la matriz de confusión.","metadata":{}},{"cell_type":"code","source":"from sklearn import metrics\n\nfpr, tpr, tr = metrics.roc_curve(true_labels, predict)\nauc = metrics.roc_auc_score(true_labels, predict)\nplt.plot(fpr,tpr,'b',label=\"AUC=\"+str(auc))\nplt.plot([0,1],[0,1],'k--')\nplt.title('Test evaluation')\nplt.grid()\nplt.legend(loc='best')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:47:40.250714Z","iopub.status.idle":"2021-11-24T22:47:40.251200Z","shell.execute_reply.started":"2021-11-24T22:47:40.250958Z","shell.execute_reply":"2021-11-24T22:47:40.250980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ncm = [eval_metrics['tn'], eval_metrics['fp'], eval_metrics['fn'], eval_metrics['tp']]\n\ngroup_names = ['True Neg','False Pos','False Neg','True Pos']\ngroup_counts = [\"{0:0.0f}\".format(value) for value in cm]\ngroup_percentages = [\"{0:.2%}\".format(value) for value in cm/np.sum(cm)]\nlabels = [f\"{v1}\\n{v2}\\n{v3}\" for v1, v2, v3 in zip(group_names,group_counts,group_percentages)]\nlabels = np.asarray(labels).reshape(2,2)\n\nsns.heatmap([[cm[0], cm[1]], [cm[2], cm[3]]], annot=labels, fmt='', cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2021-11-24T22:47:40.252639Z","iopub.status.idle":"2021-11-24T22:47:40.253411Z","shell.execute_reply.started":"2021-11-24T22:47:40.253164Z","shell.execute_reply":"2021-11-24T22:47:40.253192Z"},"trusted":true},"execution_count":null,"outputs":[]}]}