{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Detección histopatológica del cáncer.**","metadata":{}},{"cell_type":"markdown","source":"El desafío de la competencia consta de crear un algoritmo para identificar el cáncer metastásico en pequeños parches de imágenes tomadas de exploraciones patológicas digitales más grandes.","metadata":{}},{"cell_type":"markdown","source":"# **Exploración del set de datos.**","metadata":{}},{"cell_type":"markdown","source":"Características:\n1. Las imágenes tienen un tamaño de 96x96 pixeles.\n2. Una etiqueta positiva indica que la región central de 32x32 pixeles de un parche contiene al menos un píxel de tejido tumoral. El tejido tumoral en la región externa del parche no influye en la etiqueta.","metadata":{}},{"cell_type":"code","source":"import csv\nimport numpy as np\nimport pandas as pd\nimport numpy as np\nimport os","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:32:18.358642Z","iopub.execute_input":"2021-07-13T17:32:18.359076Z","iopub.status.idle":"2021-07-13T17:32:18.368983Z","shell.execute_reply.started":"2021-07-13T17:32:18.358984Z","shell.execute_reply":"2021-07-13T17:32:18.368125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ndataframe.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:32:18.380695Z","iopub.execute_input":"2021-07-13T17:32:18.381237Z","iopub.status.idle":"2021-07-13T17:32:18.964227Z","shell.execute_reply.started":"2021-07-13T17:32:18.381191Z","shell.execute_reply":"2021-07-13T17:32:18.963148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Cantidad de imágenes en el set de datos, sin patologías (0) y patológicas (1):\ndataframe['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:32:18.965762Z","iopub.execute_input":"2021-07-13T17:32:18.966077Z","iopub.status.idle":"2021-07-13T17:32:18.979391Z","shell.execute_reply.started":"2021-07-13T17:32:18.966046Z","shell.execute_reply":"2021-07-13T17:32:18.978204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Imágenes para entrenamiento\nprint(len(os.listdir('../input/histopathologic-cancer-detection/train')))\n\n#Imágenes para testeo\nprint(len(os.listdir('../input/histopathologic-cancer-detection/test')))","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:32:18.984726Z","iopub.execute_input":"2021-07-13T17:32:18.985007Z","iopub.status.idle":"2021-07-13T17:32:23.808181Z","shell.execute_reply.started":"2021-07-13T17:32:18.984980Z","shell.execute_reply":"2021-07-13T17:32:23.807365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histograma de la cantidad de imágenes por clase.\nimport matplotlib.pyplot as plt\n\nclases = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv', index_col=0)\nplt.xlabel(\"No patológicas - Patológicas.\")\nplt.ylabel(\"Cantidad de imágenes\")\nplt.hist(clases['label'], 3, color=\"blue\", ec='black')","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:32:23.809286Z","iopub.execute_input":"2021-07-13T17:32:23.809784Z","iopub.status.idle":"2021-07-13T17:32:24.324606Z","shell.execute_reply.started":"2021-07-13T17:32:23.809738Z","shell.execute_reply":"2021-07-13T17:32:24.323591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Modelo de clasificación con redes totalmente conectadas.**","metadata":{}},{"cell_type":"code","source":"from __future__ import absolute_import, division, print_function, unicode_literals\ntry:\n  %tensorflow_version 2.x\nexcept Exception:\n  pass\n\n# TensorFlow y tf.keras\nimport tensorflow as tf\nfrom tensorflow import keras\n\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:32:24.325919Z","iopub.execute_input":"2021-07-13T17:32:24.326231Z","iopub.status.idle":"2021-07-13T17:32:31.172754Z","shell.execute_reply.started":"2021-07-13T17:32:24.326200Z","shell.execute_reply":"2021-07-13T17:32:31.171636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2 \n\ndataset_train = []\nfor i in os.listdir('../input/histopathologic-cancer-detection/train/')[0:5000]:\n    dataset_train.append(cv2.imread('../input/histopathologic-cancer-detection/train/'+i, cv2.IMREAD_GRAYSCALE)/256.)\n\ndataset_train = np.array(dataset_train)\n\nprint(dataset_train.shape)\nprint(dataset_train[0])","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:44:12.948042Z","iopub.execute_input":"2021-07-13T17:44:12.948490Z","iopub.status.idle":"2021-07-13T17:44:21.181211Z","shell.execute_reply.started":"2021-07-13T17:44:12.948455Z","shell.execute_reply":"2021-07-13T17:44:21.180162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train_labels = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv')[0:5000]\ndataset_train_labels = np.array(dataset_train_labels)\n\nprint(dataset_train_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:48:06.462989Z","iopub.execute_input":"2021-07-13T17:48:06.463353Z","iopub.status.idle":"2021-07-13T17:48:06.725381Z","shell.execute_reply.started":"2021-07-13T17:48:06.463323Z","shell.execute_reply":"2021-07-13T17:48:06.724273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# División del set de entrenamiento:\nimg_train = []\nimg_test = []\n\nlong = (0.9*len(dataset_train))\n\nfor i in range(len(dataset_train)):\n  if i < long:\n    img_train.append((dataset_train[i]))\n  else:\n    img_test.append(dataset_train[i])\n\nimg_train = np.array(img_train).astype('float32')\nimg_test = np.array(img_test).astype('float32')\n    \nprint(img_train.shape)\nprint(img_test.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:48:07.749995Z","iopub.execute_input":"2021-07-13T17:48:07.750388Z","iopub.status.idle":"2021-07-13T17:48:07.948284Z","shell.execute_reply.started":"2021-07-13T17:48:07.750352Z","shell.execute_reply":"2021-07-13T17:48:07.947183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# División de las etiquetas: \ntrain_labels = []\ntest_labels = []\n\nlong1 = (0.9*len(dataset_train_labels))\n\nfor i in range(len(dataset_train_labels)):\n  if i < long1:\n    train_labels.append(dataset_train_labels[i][1])\n  else:\n    test_labels.append(dataset_train_labels[i][1])\n        \ntrain_labels = np.array(train_labels).astype('float32')\ntest_labels = np.array(test_labels).astype('float32')\n\nprint(train_labels.shape)\nprint(test_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:48:08.180792Z","iopub.execute_input":"2021-07-13T17:48:08.181197Z","iopub.status.idle":"2021-07-13T17:48:08.195659Z","shell.execute_reply.started":"2021-07-13T17:48:08.181156Z","shell.execute_reply":"2021-07-13T17:48:08.194343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.Sequential([\n    keras.layers.Flatten(input_shape=(96, 96)),\n    keras.layers.Dense(128, activation='relu'),\n    keras.layers.Dense(64, activation='relu'),\n    keras.layers.Dense(1, activation='sigmoid')\n])","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:48:09.517666Z","iopub.execute_input":"2021-07-13T17:48:09.518058Z","iopub.status.idle":"2021-07-13T17:48:09.561530Z","shell.execute_reply.started":"2021-07-13T17:48:09.518021Z","shell.execute_reply":"2021-07-13T17:48:09.560566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='sgd',\n              loss='binary_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:48:10.460231Z","iopub.execute_input":"2021-07-13T17:48:10.460581Z","iopub.status.idle":"2021-07-13T17:48:10.473603Z","shell.execute_reply.started":"2021-07-13T17:48:10.460553Z","shell.execute_reply":"2021-07-13T17:48:10.472492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Entrenamiento del modelo\nmodel.fit(img_train, train_labels, epochs = 10)","metadata":{"execution":{"iopub.status.busy":"2021-07-13T17:48:11.115461Z","iopub.execute_input":"2021-07-13T17:48:11.115848Z","iopub.status.idle":"2021-07-13T17:48:18.894785Z","shell.execute_reply.started":"2021-07-13T17:48:11.115816Z","shell.execute_reply":"2021-07-13T17:48:18.893712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}