{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nbase_dir= '../input/histopathologic-cancer-detection/'\nos.listdir(base_dir)\n\nimport cv2\n\nimport matplotlib.pyplot as plt\n\nfrom time import time\n\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.vgg16 import VGG16  \n#VGG16 is a convolution neural net (CNN ) architecture which was used to win ILSVR(Imagenet) competition in 2014. It is considered to be one of the excellent\n#vision model architecture till date. Most unique thing about VGG16 is that instead of having a large number of hyper-parameter \n#they focused on having convolution layers of 3x3 filter with a stride 1 and always used same padding and maxpool layer of 2x2 filter of stride 2. \n#It follows this arrangement of convolution and max pool layers consistently throughout the whole architecture. \n#In the end it has 2 FC(fully connected layers) followed by a softmax for output. \n#The 16 in VGG16 refers to it has 16 layers that have weights. This network is a pretty large network and it has about 138 million (approx) parameters.\n\nfrom keras.callbacks import TensorBoard\n#TensorBoard is a visualization tool provided with TensorFlow. This callback logs events for TensorBoard, including:\n#Metrics summary plots, Training graph visualization, Activation histograms, Sampled profiling\n\n#You can use callbacks to:\n#Write TensorBoard logs after every batch of training to monitor your metrics, Periodically save your model to disk, Do early stopping, Get a view on internal states and statistics of a model during training\n\nfrom keras.models import Sequential\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.layers.convolutional import Conv2D\nfrom keras.layers.convolutional import MaxPooling2D\nfrom keras.layers.core import Activation\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Dense\nfrom keras import backend as K\n\n#If our training is bouncing a lot on epochs then we need to decrease the learning rate so that we can reach global minima.\n#ModelCheckpoint helps us to save the model by monitoring a specific parameter of the model. \n#In this case I am monitoring validation accuracy by passing val_acc to ModelCheckpoint. \n#The model will only be saved to disk if the validation accuracy of the model in current epoch is greater than what it was in the last epoch.\n#EarlyStopping helps us to stop the training of the model early if there is no increase in the parameter which I have set to monitor in EarlyStopping. \n#In this case I am monitoring validation accuracy by passing val_acc to EarlyStopping. \n#I have here set patience to 20 which means that the model will stop to train if it doesn’t see any rise in validation accuracy in 20 epochs.\n\n#from keras.callbacks import ModelCheckpoint, EarlyStopping\n\n#checkpoint = ModelCheckpoint(\"vgg16_1.h5\", monitor='val_acc', verbose=1, save_best_only=True, save_weights_only=False, mode='auto', period=1)\n#early = EarlyStopping(monitor='val_acc', min_delta=0, patience=20, verbose=1, mode='auto')\n#hist = model.fit_generator(steps_per_epoch=100,generator=traindata, validation_data= testdata, validation_steps=10,epochs=100,callbacks=[checkpoint,early])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img= plt.imread('/kaggle/input/histopathologic-cancer-detection/train/'+ df.iloc[0]['id']+'.tif')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('No of Images: ' ,len(df))\nprint('% Images with label1:' ,round(len(df[df['label']==1])/len(df)*100,3), '%')\nprint('% Images with label0:' , round(len(df[df['label']==0])/len(df)*100,3), \"%\")\nprint('Image shape :' ,img.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in (np.arange(10,15)):\n    img= plt.imread('/kaggle/input/histopathologic-cancer-detection/train/'+ df.iloc[i]['id']+ '.tif')\n    print(df.iloc[i]['label'])\n    plt.imshow(img)\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n\ndef append_ext(fn):\n    return fn+\".tif\"\ndf[\"id\"]=df[\"id\"].apply(append_ext)\n\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen= ImageDataGenerator(horizontal_flip=True, vertical_flip=True, rotation_range=15, rescale=1./255, validation_split=0.15)\ntest_datagen= ImageDataGenerator(rescale=1./255)\n\ntrain_path= '/kaggle/input/histopathologic-cancer-detection/train/'\nval_path= '/kaggle/input/histopathologic-cancer-detection/train/'\n\ntrain_generator= train_datagen.flow_from_dataframe(dataframe=df, directory=train_path,\n                                                  x_col='id', y_col='label',\n                                                  subset='training', batch_size=64, class_mode='binary', #Mode for yielding the targets: - \"binary\": 1D numpy array of binary labels\n                                                  target_size= (96,96)) #tuple of integers (height, width), default: (256, 256). The dimensions to which all images found will be resized\n\nvalidation_generator= train_datagen.flow_from_dataframe(dataframe=df, directory=val_path, target_size= (96,96), subset='validation', shuffle=False,\n                                                       x_col='id', y_col='label', batch_size=64,class_mode='binary')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Sequential()\nmodel.add(Conv2D(filters=16, kernel_size=3, padding='same', activation='relu', input_shape=(96,96,3)))\nmodel.add(Conv2D(filters=16, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Conv2D(filters=16, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(MaxPooling2D(pool_size=3))\n\nmodel.add(Conv2D(filters=32, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Conv2D(filters=32, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Conv2D(filters = 32, kernel_size = 3, padding ='same', activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(MaxPooling2D(pool_size=3))\n\nmodel.add(Conv2D(filters=64, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Conv2D(filters=64, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Conv2D(filters=64, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(MaxPooling2D(pool_size=3))\n\nmodel.add(Conv2D(filters=128, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Conv2D(filters=128, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Conv2D(filters=128, kernel_size=3, padding='same', activation='relu'))\nmodel.add(Dropout(0.3))\n\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nstep_size_train= train_generator.n//train_generator.batch_size\nstep_size_val= validation_generator.n//validation_generator.batch_size","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit_generator(train_generator, steps_per_epoch= step_size_train, validation_data= validation_generator, validation_steps= step_size_val, epochs=5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/input/histopathologic-cancer-detection/test')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.predict(test_df[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df= os.listdir('/kaggle/input/histopathologic-cancer-detection/test')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img= plt.imread('/kaggle/input/histopathologic-cancer-detection/test/5fde41ce8c6048a5c2f38eca12d6528fa312cdbb.tif')\nplt.imshow(img)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a= np.array(img/255.)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\n\nfrom matplotlib.pyplot import imread\n\n# Kaggle testing\nfrom glob import glob\nTESTING_BATCH_SIZE = 64\ntesting_files = glob(os.path.join('/kaggle/input/histopathologic-cancer-detection/test','*.tif'))\nsubmission = pd.DataFrame()\nprint(len(testing_files))\nfor index in range(0, len(testing_files), TESTING_BATCH_SIZE):\n    data_frame = pd.DataFrame({'path': testing_files[index:index+TESTING_BATCH_SIZE]})\n    data_frame['id'] = data_frame.path.map(lambda x: x.split('/')[3].split(\".\")[0])\n    data_frame['image'] = data_frame['path'].map(imread)\n    images = np.stack(data_frame.image, axis=0)\n    predicted_labels = [model.predict(np.expand_dims(image/255.0, axis=0))[0][0] for image in images]\n    predictions = np.array(predicted_labels)\n    data_frame['label'] = predictions\n    submission = pd.concat([submission, data_frame[[\"id\", \"label\"]]])\n    if index % 1000 == 0 :\n        print(index/len(testing_files) * 100)\nsubmission.to_csv('submission_new_model.csv', index=False, header=True)\nprint(submission.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}