{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n\n\n\n# # confirm TensorFlow sees the GPU\n# from tensorflow.python.client import device_lib\n# assert 'GPU' in str(device_lib.list_local_devices())\n\n# confirm Keras sees the GPU (for TensorFlow 1.X + Keras)\n# from keras import backend\n# assert len(backend.tensorflow_backend._get_available_gpus()) > 0\n\n\n# from __future__ import print_function, division\nfrom builtins import range, input\n# Note: you may need to update your version of future\n# sudo pip install -U future\nimport tensorflow as tf\n# import tensorflow.compat.v1 as tf\nfrom keras.layers import Input, Lambda, Dense, Flatten\nfrom keras.models import Model\nfrom keras import backend\nfrom keras.applications.vgg16 import VGG16\nfrom keras.applications.vgg16 import preprocess_input\nfrom keras.preprocessing import image\nfrom keras.preprocessing.image import ImageDataGenerator\n\nfrom sklearn.metrics import confusion_matrix\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n\nfrom glob import glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make sure GPU is available\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))\nprint(\"Num CPUs Available: \", len(tf.config.list_physical_devices('CPU')))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label dataframe inspection\ntrain_labels = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\ntrain_labels.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define train/test path\ntrain_path = '../input/histopathologic-cancer-detection/train'\nvalid_path = '../input/histopathologic-cancer-detection/test'\n\n# useful for getting number of files\nimage_files = glob(train_path + '/*.tif')\nvalid_image_files = glob(valid_path + '/*.tif')\n\n# image snapshot\nfor i in range(5):\n    _label = 0\n    image_path = np.random.choice(train_labels[train_labels['label'] == _label]['id'].apply(lambda x: train_path + \"/\" + x + \".tif\"))\n    #print(image_path)\n    plt.title(f\"Path: {image_path} | label : {_label}\")\n    plt.imshow(image.load_img(image_path))\n    \n    plt.show()\n    \nfor i in range(5):\n    _label = 1\n    image_path = np.random.choice(train_labels[train_labels['label'] == _label]['id'].apply(lambda x: train_path + \"/\" + x + \".tif\"))\n    #print(image_path)\n    plt.title(f\"Path: {image_path} | label : {_label}\")\n    plt.imshow(image.load_img(image_path))\n    \n    plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# inspect csv ","metadata":{}},{"cell_type":"code","source":"# !mkdir ./train_set\n# !mkdir ./train_set/label_0\n# !mkdir ./train_set/label_1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !ls train_set/\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_path = '../input/histopathologic-cancer-detection/train'\n# valid_path = '../input/histopathologic-cancer-detection/test'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_set_label = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\n# train_set_label[train_set_label['label'] == 1].head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# import shutil\n\n# for index, row in train_set_label[train_set_label['label'] == 1].iterrows():\n#     img_path = f\"../input/histopathologic-cancer-detection/train/{row['id']}.tif\"\n#     print(index, img_path)\n#     shutil.copy2(img_path, f\"/kaggle/working/label_1\") # complete target filename given\n\n# for index, row in train_set_label[train_set_label['label'] == 0].iterrows():\n#     img_path = f\"../input/histopathologic-cancer-detection/train/{row['id']}.tif\"\n#     print(index, img_path)\n#     shutil.copy2(img_path, f\"/kaggle/working/label_0\") # complete target filename given\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_files = glob(train_path + '/*tif')\n# len(image_files)\n# # print(train_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !ls ./train_set/label_1/\n# !pwd ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Use pretrained model","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = tf.compat.v1.ConfigProto( device_count = {'GPU': 1 , 'CPU': 1} ) \ntf.compat.v1.Session()\nsess = tf.compat.v1.Session(config=config)\n# sess = tf.Session(config=config) \n# backend.set_session(sess)\nsess = tf.compat.v1.keras.backend.get_session()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print hardware spec.\nfrom tensorflow.python.client import device_lib\nprint(device_lib.list_local_devices())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras==2.1 -i https://pypi.douban.com/simple/","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from keras import backend as K\n# # tf.compat.v1.keras.backend.\n# # K.tensorflow_backend._get_available_gpus()\n# # K.tensorflow_backend._get_available_gpus()\n\n# dir(K)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ntrain_labels = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\ntrain_labels['id'] = train_labels['id'].apply(lambda x: '{}.tif'.format(x))\n\n# Data augmentation pipeline\ntrain_datagen = tf.keras.preprocessing.image.ImageDataGenerator()\n# Reading files from path in data frame\ntrain_ds = train_datagen.flow_from_dataframe(train_labels,directory = '../input/histopathologic-cancer-detection/train', x_col = 'id', y_col = 'label', class_mode='raw')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from __future__ import print_function, division\nfrom builtins import range, input\n# Note: you may need to update your version of future\n# sudo pip install -U future\n\nfrom keras.layers import Input, Lambda, Dense, Flatten\nfrom keras.models import Model\nfrom keras.applications.resnet50 import ResNet50, preprocess_input\n# from keras.applications.inception_v3 import InceptionV3, preprocess_input\nfrom keras.preprocessing import image\nfrom keras.preprocessing.image import ImageDataGenerator\n\nfrom sklearn.metrics import confusion_matrix\nimport numpy as np\n\n\nfrom glob import glob\n\n\n# re-size all the images to this\nIMAGE_SIZE = [96, 96] # \n\n# training config:\nepochs = 16\nbatch_size = 32\n\n# # https://www.kaggle.com/paultimothymooney/blood-cells\n# train_path = '../input/histopathologic-cancer-detection/train'\n# valid_path = '../input/histopathologic-cancer-detection/test'\n\n# # https://www.kaggle.com/moltean/fruits\n# # train_path = '../large_files/fruits-360/Training'\n# # valid_path = '../large_files/fruits-360/Validation'\n# # train_path = '../large_files/fruits-360-small/Training'\n# # valid_path = '../large_files/fruits-360-small/Validation'\n\n# # useful for getting number of files\n# image_files = glob(train_path + '/*.tif')\n# valid_image_files = glob(valid_path + '/*.tif')\n\n# # useful for getting number of classes\n# # folders = glob(train_path + '/*')\n\n\n# # # look at an image \n# # for i in range(3):\n# #     plt.imshow(image.load_img(np.random.choice(image_files)))\n# #     plt.show()\n\n# # add preprocessing layer to the front of VGG\n\n\n\n\n\n\n\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res = ResNet50(input_shape=IMAGE_SIZE + [3], weights='imagenet', include_top=False)\n\nfor layer in res.layers:\n    layer.trainable = False\n\n# our layers - you can add more if you want\nx = Flatten()(res.output)\nx = Dense(1000, activation='relu')(x)\n#prediction = Dense(2, activation='softmax')(x)\n# prediction = Dense(2, activation='sigmoid')(x)\nprediction = Dense(1, activation='sigmoid')(x)\n# model.add(layers.Dense(1, activation='sigmoid'))\n# create a model object\nmodel = Model(inputs=res.input, outputs=prediction)\n\n\n# tell the model what cost and optimization method to use\nmodel.compile(\n  #loss='categorical_crossentropy',\n    loss='binary_crossentropy',\n    optimizer='rmsprop',\n    metrics=['accuracy']\n)\n\n\n\n# image-label mapping\ndf = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\ndf['id'] = df['id'].apply(lambda x: '{}.tif'.format(x))\n\n# Data augmentation pipeline\n# train_datagen = tf.keras.preprocessing.image.ImageDataGenerator()\n# create an instance of ImageDataGenerator\ntrain_datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    preprocessing_function=preprocess_input,\n    validation_split=0.2\n)\n# Reading files from path in data frame\ntrain_generator = train_datagen.flow_from_dataframe(\n        dataframe = df,\n        directory = '../input/histopathologic-cancer-detection/train', \n        x_col = 'id',\n        y_col = 'label', \n        class_mode='raw',\n        target_size=IMAGE_SIZE,\n        subset='training' # set as training data\n)\n\nvalidation_generator = train_datagen.flow_from_dataframe(\n        dataframe = df,\n        directory = '../input/histopathologic-cancer-detection/train', \n        x_col = 'id',\n        y_col = 'label', \n        class_mode='raw',\n        target_size=IMAGE_SIZE,\n        subset='validation' # set as validation data\n)\n\n# train_generator = train_datagen.flow_from_directory(\n#     train_data_dir,\n#     target_size=(img_height, img_width),\n#     batch_size=batch_size,\n#     class_mode='binary',\n#     subset='training') # set as training data\n\n# validation_generator = train_datagen.flow_from_directory(\n#     train_data_dir, # same directory as training data\n#     target_size=(img_height, img_width),\n#     batch_size=batch_size,\n#     class_mode='binary',\n#     subset='validation') # set as validation data\n\n# fit the model\nr = model.fit(\n    train_generator,\n    steps_per_epoch = train_generator.samples // batch_size,\n    validation_data = validation_generator, \n    validation_steps = validation_generator.samples // batch_size,\n    epochs = epochs\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# view the structure of the model\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot loss per iteration\nimport matplotlib.pyplot as plt\nplt.plot(r.history['loss'], label='loss')\nplt.plot(r.history['val_loss'], label='val_loss')\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot accuracy per iteration\nplt.plot(r.history['accuracy'], label='acc')\nplt.plot(r.history['val_accuracy'], label='val_acc')\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save trained model\nmodel.save_weights('roger_trained_model_histopathologic_cancer_detection_checkpoint.h5')\n# ../input/histopathologic-cancer-detection","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['id'] = train_labels['id'].apply(lambda x: '{}.tif'.format(x))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction (zero matching rate)   ><\"  sosad !!","metadata":{}},{"cell_type":"code","source":"#for date, row in train_labels[train_labels['label'] == 1].head(10000).iterrows():\nfor date, row in train_labels.head(100).iterrows():\n    #print(f\"{row['id']} | label: {row['label']}\")\n    \n    image_path = f\"../input/histopathologic-cancer-detection/train/{row['id']}\"\n    #print(image_path)\n    \n\n    input_arr = tf.keras.preprocessing.image.img_to_array(image.load_img(image_path, target_size=[96,96],color_mode='rgb' ))\n#     input_arr = cv2.imread(image_path)\n    input_arr2 = np.array([input_arr])\n    input_arr3 = input_arr2.astype('float32') / 255\n    \n    predictions = model.predict(input_arr3)\n    predicted = np.argmax(predictions, axis=-1)[0]\n    \n    print(f\"{row['id']} | label: {row['label']} | predicted: {predicted}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}