{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Detección histopatológica del cáncer.**","metadata":{"papermill":{"duration":0.020404,"end_time":"2021-07-25T19:28:56.296395","exception":false,"start_time":"2021-07-25T19:28:56.275991","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"El desafío de la competencia consta de crear un algoritmo para identificar el cáncer metastásico en pequeños parches de imágenes tomadas de exploraciones patológicas digitales más grandes.","metadata":{"papermill":{"duration":0.019166,"end_time":"2021-07-25T19:28:56.335092","exception":false,"start_time":"2021-07-25T19:28:56.315926","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# **Exploración del set de datos.**","metadata":{"papermill":{"duration":0.018714,"end_time":"2021-07-25T19:28:56.373207","exception":false,"start_time":"2021-07-25T19:28:56.354493","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Características:\n1. Las imágenes tienen un tamaño de 96x96 pixeles.\n2. Una etiqueta positiva indica que la región central de 32x32 pixeles de un parche contiene al menos un píxel de tejido tumoral. El tejido tumoral en la región externa del parche no influye en la etiqueta.","metadata":{"papermill":{"duration":0.019178,"end_time":"2021-07-25T19:28:56.411996","exception":false,"start_time":"2021-07-25T19:28:56.392818","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import cv2\nimport csv\nimport numpy as np\nimport pandas as pd\nimport numpy as np\nimport os","metadata":{"papermill":{"duration":0.259165,"end_time":"2021-07-25T19:28:56.690268","exception":false,"start_time":"2021-07-25T19:28:56.431103","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:08.490539Z","iopub.execute_input":"2021-07-27T14:26:08.491137Z","iopub.status.idle":"2021-07-27T14:26:08.655996Z","shell.execute_reply.started":"2021-07-27T14:26:08.491044Z","shell.execute_reply":"2021-07-27T14:26:08.655279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ndataframe.head()","metadata":{"papermill":{"duration":0.577467,"end_time":"2021-07-25T19:28:57.287641","exception":false,"start_time":"2021-07-25T19:28:56.710174","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:08.742333Z","iopub.execute_input":"2021-07-27T14:26:08.742817Z","iopub.status.idle":"2021-07-27T14:26:09.184078Z","shell.execute_reply.started":"2021-07-27T14:26:08.742785Z","shell.execute_reply":"2021-07-27T14:26:09.183294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe.info()","metadata":{"papermill":{"duration":0.068615,"end_time":"2021-07-25T19:28:57.376861","exception":false,"start_time":"2021-07-25T19:28:57.308246","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:09.185392Z","iopub.execute_input":"2021-07-27T14:26:09.185844Z","iopub.status.idle":"2021-07-27T14:26:09.231713Z","shell.execute_reply.started":"2021-07-27T14:26:09.185813Z","shell.execute_reply":"2021-07-27T14:26:09.230564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe.sort_values(by='id', inplace = True)\ndataframe.head()","metadata":{"papermill":{"duration":0.554219,"end_time":"2021-07-25T19:28:57.952486","exception":false,"start_time":"2021-07-25T19:28:57.398267","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:09.234229Z","iopub.execute_input":"2021-07-27T14:26:09.2347Z","iopub.status.idle":"2021-07-27T14:26:09.760134Z","shell.execute_reply.started":"2021-07-27T14:26:09.234653Z","shell.execute_reply":"2021-07-27T14:26:09.759022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train_labels = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv')\ndataset_train_labels.sort_values(by='id', inplace = True)\n\ndataset_train_labels.head()","metadata":{"papermill":{"duration":0.79115,"end_time":"2021-07-25T19:28:58.769736","exception":false,"start_time":"2021-07-25T19:28:57.978586","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:09.762109Z","iopub.execute_input":"2021-07-27T14:26:09.762829Z","iopub.status.idle":"2021-07-27T14:26:10.576412Z","shell.execute_reply.started":"2021-07-27T14:26:09.762779Z","shell.execute_reply":"2021-07-27T14:26:10.575623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Cantidad de imágenes en el set de datos, sin patologías (0) y patológicas (1):\ndataframe['label'].value_counts()","metadata":{"papermill":{"duration":0.03435,"end_time":"2021-07-25T19:28:58.826064","exception":false,"start_time":"2021-07-25T19:28:58.791714","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:10.577847Z","iopub.execute_input":"2021-07-27T14:26:10.578283Z","iopub.status.idle":"2021-07-27T14:26:10.588217Z","shell.execute_reply.started":"2021-07-27T14:26:10.578251Z","shell.execute_reply":"2021-07-27T14:26:10.586994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Imágenes para entrenamiento\nprint('Cantidad de imágenes para el entrenamiento:')\nprint(len(os.listdir('../input/histopathologic-cancer-detection/train')))\n\n#Imágenes para testeo\nprint('Cantidad de imágenes para testeo:')\nprint(len(os.listdir('../input/histopathologic-cancer-detection/test')))","metadata":{"papermill":{"duration":5.387328,"end_time":"2021-07-25T19:29:04.234642","exception":false,"start_time":"2021-07-25T19:28:58.847314","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:10.590604Z","iopub.execute_input":"2021-07-27T14:26:10.590977Z","iopub.status.idle":"2021-07-27T14:26:16.865311Z","shell.execute_reply.started":"2021-07-27T14:26:10.590944Z","shell.execute_reply":"2021-07-27T14:26:16.863735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histograma de la cantidad de imágenes por clase.\nimport matplotlib.pyplot as plt\n\nclases = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv', index_col=0)\nplt.xlabel(\"No patológicas - Patológicas.\")\nplt.ylabel(\"Cantidad de imágenes\")\nplt.hist(clases['label'], 3, color=\"blue\", ec='black')","metadata":{"papermill":{"duration":0.506104,"end_time":"2021-07-25T19:29:04.763085","exception":false,"start_time":"2021-07-25T19:29:04.256981","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:16.867098Z","iopub.execute_input":"2021-07-27T14:26:16.867453Z","iopub.status.idle":"2021-07-27T14:26:17.445411Z","shell.execute_reply.started":"2021-07-27T14:26:16.867421Z","shell.execute_reply":"2021-07-27T14:26:17.444483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# source: https://www.kaggle.com/gpreda/honey-bee-subspecies-classification\n\ndef draw_category_images(col_name,figure_cols, df, IMAGE_PATH):\n    \n    \"\"\"\n    Give a column in a dataframe,\n    this function takes a sample of each class and displays that\n    sample on one row. The sample size is the same as figure_cols which\n    is the number of columns in the figure.\n    Because this function takes a random sample, each time the function is run it\n    displays different images.\n    \"\"\"\n    \n\n    categories = (df.groupby([col_name])[col_name].nunique()).index\n    f, ax = plt.subplots(nrows=len(categories),ncols=figure_cols, \n                         figsize=(4*figure_cols,4*len(categories))) # adjust size here\n    # draw a number of images for each location\n    for i, cat in enumerate(categories):\n        sample = df[df[col_name]==cat].sample(figure_cols) # figure_cols is also the sample size\n        for j in range(0,figure_cols):\n            file=IMAGE_PATH + sample.iloc[j]['id'] + '.tif'\n            im=cv2.imread(file)\n            ax[i, j].imshow(im, resample=True, cmap='gray')\n            ax[i, j].set_title(cat, fontsize=16)  \n    plt.tight_layout()\n    plt.show()\nIMAGE_PATH = ('../input/histopathologic-cancer-detection/train/')\n\ndraw_category_images('label',4, dataframe, IMAGE_PATH)","metadata":{"papermill":{"duration":1.519944,"end_time":"2021-07-25T19:29:06.307013","exception":false,"start_time":"2021-07-25T19:29:04.787069","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:17.446583Z","iopub.execute_input":"2021-07-27T14:26:17.447052Z","iopub.status.idle":"2021-07-27T14:26:19.040094Z","shell.execute_reply.started":"2021-07-27T14:26:17.447021Z","shell.execute_reply":"2021-07-27T14:26:19.038401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Modelo convolucional.**","metadata":{"papermill":{"duration":0.052545,"end_time":"2021-07-25T19:29:06.413099","exception":false,"start_time":"2021-07-25T19:29:06.360554","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from __future__ import absolute_import, division, print_function, unicode_literals\ntry:\n  %tensorflow_version 2.x\nexcept Exception:\n  pass\n\n# TensorFlow y tf.keras\nimport tensorflow as tf\nfrom tensorflow import keras\n\nprint(tf.__version__)","metadata":{"papermill":{"duration":6.279011,"end_time":"2021-07-25T19:29:12.746065","exception":false,"start_time":"2021-07-25T19:29:06.467054","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:19.042358Z","iopub.execute_input":"2021-07-27T14:26:19.043246Z","iopub.status.idle":"2021-07-27T14:26:26.400648Z","shell.execute_reply.started":"2021-07-27T14:26:19.04318Z","shell.execute_reply":"2021-07-27T14:26:26.399304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = os.listdir('../input/histopathologic-cancer-detection/train/')\nfiles.sort()\n\ndataset_train = []\n\nfor i in os.listdir('../input/histopathologic-cancer-detection/train/')[0:1000]:\n    dataset_train.append(cv2.imread('../input/histopathologic-cancer-detection/train/'+i, cv2.IMREAD_GRAYSCALE)/256.)\n\ndataset_train = np.array(dataset_train)\nprint(dataset_train.shape)","metadata":{"papermill":{"duration":636.502447,"end_time":"2021-07-25T19:39:49.302942","exception":false,"start_time":"2021-07-25T19:29:12.800495","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:26.40466Z","iopub.execute_input":"2021-07-27T14:26:26.405131Z","iopub.status.idle":"2021-07-27T14:26:35.185787Z","shell.execute_reply.started":"2021-07-27T14:26:26.405082Z","shell.execute_reply":"2021-07-27T14:26:35.18474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train_labels = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv')[0:1000]\ndataset_train_labels.sort_values(by='id', inplace = True)\n\ndataset_train_labels = np.array(dataset_train_labels)\n\nprint(dataset_train_labels.shape)","metadata":{"papermill":{"duration":0.619331,"end_time":"2021-07-25T19:39:49.977734","exception":false,"start_time":"2021-07-25T19:39:49.358403","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:35.187311Z","iopub.execute_input":"2021-07-27T14:26:35.187613Z","iopub.status.idle":"2021-07-27T14:26:35.454761Z","shell.execute_reply.started":"2021-07-27T14:26:35.187582Z","shell.execute_reply":"2021-07-27T14:26:35.454003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# División del set de entrenamiento:\nimg_train = []\nimg_test = []\n\nlong = (0.9*len(dataset_train))\n\nfor i in range(len(dataset_train)):\n  if i < long:\n    img_train.append((dataset_train[i]))\n  else:\n    img_test.append(dataset_train[i])\n\nimg_train = np.array(img_train).astype('float32')\nimg_test = np.array(img_test).astype('float32')\n    \nprint(img_train.shape)\nprint(img_test.shape)","metadata":{"papermill":{"duration":4.415266,"end_time":"2021-07-25T19:39:54.448194","exception":false,"start_time":"2021-07-25T19:39:50.032928","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:35.455919Z","iopub.execute_input":"2021-07-27T14:26:35.456439Z","iopub.status.idle":"2021-07-27T14:26:35.516707Z","shell.execute_reply.started":"2021-07-27T14:26:35.456407Z","shell.execute_reply":"2021-07-27T14:26:35.515394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# División de las etiquetas: \ntrain_labels = []\ntest_labels = []\n\nlong1 = (0.9*len(dataset_train_labels))\n\nfor i in range(len(dataset_train_labels)):\n  if i < long1:\n    train_labels.append(dataset_train_labels[i][1])\n  else:\n    test_labels.append(dataset_train_labels[i][1])\n        \ntrain_labels = np.array(train_labels).astype('float32')\ntest_labels = np.array(test_labels).astype('float32')\n\nprint(train_labels.shape)\nprint(test_labels.shape)","metadata":{"papermill":{"duration":0.139568,"end_time":"2021-07-25T19:39:54.643463","exception":false,"start_time":"2021-07-25T19:39:54.503895","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:35.518107Z","iopub.execute_input":"2021-07-27T14:26:35.518428Z","iopub.status.idle":"2021-07-27T14:26:35.529902Z","shell.execute_reply.started":"2021-07-27T14:26:35.518399Z","shell.execute_reply":"2021-07-27T14:26:35.528556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.Sequential()\n\nmodel.add(keras.layers.Conv2D(filters=64, padding='valid', kernel_size=(3, 3), activation='relu', input_shape=(96,96,1)))\nmodel.add(keras.layers.MaxPooling2D(2, 2))\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-27T14:26:35.531497Z","iopub.execute_input":"2021-07-27T14:26:35.531813Z","iopub.status.idle":"2021-07-27T14:26:35.661552Z","shell.execute_reply.started":"2021-07-27T14:26:35.531779Z","shell.execute_reply":"2021-07-27T14:26:35.660372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.add(keras.layers.Flatten())\nmodel.add(keras.layers.Dense(120, activation='relu'))\nmodel.add(keras.layers.Dense(84, activation='relu'))\nmodel.add(keras.layers.Dense(10, activation = 'sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2021-07-27T14:26:35.662984Z","iopub.execute_input":"2021-07-27T14:26:35.663319Z","iopub.status.idle":"2021-07-27T14:26:35.858804Z","shell.execute_reply.started":"2021-07-27T14:26:35.663285Z","shell.execute_reply":"2021-07-27T14:26:35.857488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam',\n              loss='binary_crossentropy',\n              metrics=['accuracy'])","metadata":{"papermill":{"duration":0.076542,"end_time":"2021-07-25T19:39:55.056992","exception":false,"start_time":"2021-07-25T19:39:54.98045","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:35.860636Z","iopub.execute_input":"2021-07-27T14:26:35.861185Z","iopub.status.idle":"2021-07-27T14:26:35.882593Z","shell.execute_reply.started":"2021-07-27T14:26:35.861119Z","shell.execute_reply":"2021-07-27T14:26:35.88136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Entrenamiento del modelo\nmodel.fit(img_train, train_labels, epochs = 20)","metadata":{"papermill":{"duration":666.357842,"end_time":"2021-07-25T19:51:01.469523","exception":false,"start_time":"2021-07-25T19:39:55.111681","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:35.884491Z","iopub.execute_input":"2021-07-27T14:26:35.884879Z","iopub.status.idle":"2021-07-27T14:26:36.572039Z","shell.execute_reply.started":"2021-07-27T14:26:35.884843Z","shell.execute_reply":"2021-07-27T14:26:36.56923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(img_test,  test_labels, verbose=2)\n\nprint('\\nTest accuracy:', test_acc)","metadata":{"papermill":{"duration":4.976288,"end_time":"2021-07-25T19:51:10.15375","exception":false,"start_time":"2021-07-25T19:51:05.177462","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:36.573302Z","iopub.status.idle":"2021-07-27T14:26:36.573784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicts = model.predict(img_test)\nthreshold = 0.5\npredicts = (predicts >= threshold).astype(int)","metadata":{"papermill":{"duration":4.445797,"end_time":"2021-07-25T19:51:18.317488","exception":false,"start_time":"2021-07-25T19:51:13.871691","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:36.575169Z","iopub.status.idle":"2021-07-27T14:26:36.575652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.tight_layout()\n    plt.ylabel('Etiqueta correcta')\n    plt.xlabel('Etiqueta predicha')\nfrom sklearn.metrics import confusion_matrix\nimport itertools\n\ncm = confusion_matrix(test_labels, predicts)\ntn, fp, fn, tp = confusion_matrix(test_labels, predicts).ravel()\nplot_confusion_matrix(cm,[\"0\",\"1\"])","metadata":{"papermill":{"duration":4.825401,"end_time":"2021-07-25T19:51:26.879678","exception":false,"start_time":"2021-07-25T19:51:22.054277","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-27T14:26:36.576772Z","iopub.status.idle":"2021-07-27T14:26:36.577207Z"},"trusted":true},"execution_count":null,"outputs":[]}]}