{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importando librerías","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\n\nimport matplotlib.pyplot as plt\nfrom skimage.io import imread #read images from files\nimport matplotlib.patches as patches\nfrom PIL import Image\n\nimport cv2","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n","is_executing":false},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Revisando los labels de train","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv')\n\ndf['label'].value_counts()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n","is_executing":false},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Me quedo solo con 1000 filas para que el procesamiento de esto sea más rápido","metadata":{}},{"cell_type":"code","source":"df = df.sample(1000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_positive = df[df[\"label\"] == 1].sample(5)['id']\n\n\ncancer_negative = df[df[\"label\"] == 0].sample(5)['id']\n","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n","is_executing":false},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readImage(path):\n    # OpenCV reads the image in bgr format by default\n    bgr_img = cv2.imread(path)\n    # We flip it to rgb for visualization purposes\n    b,g,r = cv2.split(bgr_img)\n    rgb_img = cv2.merge([r,g,b])\n    return rgb_img\n\n","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n","is_executing":false},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfig, ax = plt.subplots(2,5, figsize=(20,8))\nfig.suptitle('Histopathologic scans of lymph node sections',fontsize=20)\npath_dir = '../input/histopathologic-cancer-detection'\n\n# Negatives\nfor i, idx in enumerate(cancer_negative):\n    path = os.path.join(path_dir, 'train', idx)\n    ax[0,i].imshow(readImage(path + '.tif'))\n    \nax[0,0].set_ylabel('Negative samples', size='large')\n\n# Positives\nfor i, idx in enumerate(cancer_positive):\n    path = os.path.join(path_dir, 'train', idx)\n    ax[1,i].imshow(readImage(path + '.tif'))\n    \nax[1,0].set_ylabel('Tumor tissue samples', size='large')","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n","is_executing":false},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Leyendo imágenes\ndf['image_path'] = path_dir + '/train/' + df['id'] + '.tif'\ndf.sample(3)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image'] = df['image_path'].map(imread)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sample(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing 96x96x3 numpy arrays with Keras\n\nTrying the code in https://keras.io/getting_started/intro_to_keras_for_engineers/#training-models-with-fit","metadata":{"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-11-04T23:13:05.796904Z","iopub.execute_input":"2022-11-04T23:13:05.797349Z","iopub.status.idle":"2022-11-04T23:13:05.803509Z","shell.execute_reply.started":"2022-11-04T23:13:05.797316Z","shell.execute_reply":"2022-11-04T23:13:05.802178Z"}}},{"cell_type":"code","source":"# We know we have images from 96x96 in rgb\ninputs = keras.Input(shape=(96, 96, 3))\n\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.layers import CenterCrop\nfrom tensorflow.keras.layers import Rescaling\n\n# Center-crop images to 96x96 just in case\nx = CenterCrop(height=96, width=96)(inputs)\n# Rescale images to [0, 1]\nx = Rescaling(scale=1.0 / 255)(x)\n\n# Apply some convolution and pooling layers\nx = layers.Conv2D(filters=32, kernel_size=(3, 3), activation=\"relu\")(x)\nx = layers.MaxPooling2D(pool_size=(3, 3))(x)\nx = layers.Conv2D(filters=32, kernel_size=(3, 3), activation=\"relu\")(x)\nx = layers.MaxPooling2D(pool_size=(3, 3))(x)\nx = layers.Conv2D(filters=32, kernel_size=(3, 3), activation=\"relu\")(x)\n\n# Apply global average pooling to get flat feature vectors\nx = layers.GlobalAveragePooling2D()(x)\n\n# Add a dense classifier on top\n# num_classes = 2\n# outputs = layers.Dense(num_classes, activation=\"softmax\")(x)\n\n# Using Dense 1 because of binary prediction\noutputs = layers.Dense(1, kernel_initializer='normal', activation='sigmoid')(x)\n# Compile model. We use the the logarithmic loss function, and the Adam gradient optimizer.\n# model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.Model(inputs=inputs, outputs=outputs)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='rmsprop', loss='binary_crossentropy', metrics=[tf.keras.metrics.BinaryAccuracy()],)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image'].shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_images = np.stack(list(df['image']), axis = 0)\ninput_images.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nfrom sklearn.preprocessing import LabelBinarizer\n\nencoder = LabelBinarizer()\ny = encoder.fit_transform(df['label'])\n\nX_train, X_val, y_train, y_val = train_test_split(input_images, y, test_size=0.2, random_state=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X_train, y_train, batch_size=32, epochs=10, \n    validation_data=(X_val, y_val),)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the test data using `evaluate`\nresults = model.evaluate(X_val, y_val, verbose=0)\nprint(\"test loss, test acc:\", results)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Generate predictions (probabilities -- the output of the last layer)\n# on new data using `predict`\nprint(\"Generate predictions for 3 samples\")\npredictions = model.predict(X_val[:3])\nprint(\"predictions shape:\", predictions.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}