{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-20T05:15:39.709571Z","iopub.execute_input":"2021-07-20T05:15:39.710267Z","iopub.status.idle":"2021-07-20T05:25:14.535651Z","shell.execute_reply.started":"2021-07-20T05:15:39.710121Z","shell.execute_reply":"2021-07-20T05:25:14.533683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport seaborn as sb\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:14.073134Z","iopub.execute_input":"2021-07-20T06:01:14.073528Z","iopub.status.idle":"2021-07-20T06:01:14.079361Z","shell.execute_reply.started":"2021-07-20T06:01:14.073494Z","shell.execute_reply":"2021-07-20T06:01:14.078462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '../input/histopathologic-cancer-detection/train/'\ntest_path = '../input/histopathologic-cancer-detection/test/'","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:19.276331Z","iopub.execute_input":"2021-07-20T06:01:19.276933Z","iopub.status.idle":"2021-07-20T06:01:19.281834Z","shell.execute_reply.started":"2021-07-20T06:01:19.27689Z","shell.execute_reply":"2021-07-20T06:01:19.280198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = os.listdir(train_path)\ntest = os.listdir(test_path)","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:20.438913Z","iopub.execute_input":"2021-07-20T06:01:20.439424Z","iopub.status.idle":"2021-07-20T06:01:25.374198Z","shell.execute_reply.started":"2021-07-20T06:01:20.439391Z","shell.execute_reply":"2021-07-20T06:01:25.373128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total no. of train images: \",len(train_path))\nprint(\"Total no. of test images: \",len(test_path))","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:25.375784Z","iopub.execute_input":"2021-07-20T06:01:25.376321Z","iopub.status.idle":"2021-07-20T06:01:25.383163Z","shell.execute_reply.started":"2021-07-20T06:01:25.376279Z","shell.execute_reply":"2021-07-20T06:01:25.381995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv(r'../input/histopathologic-cancer-detection/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:25.38523Z","iopub.execute_input":"2021-07-20T06:01:25.385601Z","iopub.status.idle":"2021-07-20T06:01:26.023404Z","shell.execute_reply.started":"2021-07-20T06:01:25.385561Z","shell.execute_reply":"2021-07-20T06:01:26.022259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:27.780885Z","iopub.execute_input":"2021-07-20T06:01:27.781295Z","iopub.status.idle":"2021-07-20T06:01:27.792055Z","shell.execute_reply.started":"2021-07-20T06:01:27.78126Z","shell.execute_reply":"2021-07-20T06:01:27.791025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train[0:5])\nprint(labels.loc[0:5])","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:28.984224Z","iopub.execute_input":"2021-07-20T06:01:28.984834Z","iopub.status.idle":"2021-07-20T06:01:29.005921Z","shell.execute_reply.started":"2021-07-20T06:01:28.984785Z","shell.execute_reply":"2021-07-20T06:01:29.004321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sb.countplot('label',data = labels).set_title('Class labels')","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:29.968381Z","iopub.execute_input":"2021-07-20T06:01:29.968848Z","iopub.status.idle":"2021-07-20T06:01:30.189285Z","shell.execute_reply.started":"2021-07-20T06:01:29.968807Z","shell.execute_reply":"2021-07-20T06:01:30.18791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(25, 6))\n# display 20 images\nfor idx, img in enumerate(np.random.choice(train, 20)):\n    ax = fig.add_subplot(2, 20//2, idx+1, xticks=[], yticks=[])\n    im = plt.imread(f'{train_path}' + img)\n    lab = labels.loc[labels['id'] == img.split('.')[0], 'label'].values[0]\n    ax.set_title(f'Label: {lab}')\n    ax.set_xlabel(im.shape[0])\n    ax.set_ylabel(im.shape[1])\n    plt.imshow(im)","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:31.116497Z","iopub.execute_input":"2021-07-20T06:01:31.116894Z","iopub.status.idle":"2021-07-20T06:01:33.542143Z","shell.execute_reply.started":"2021-07-20T06:01:31.116861Z","shell.execute_reply":"2021-07-20T06:01:33.540367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:01:36.457576Z","iopub.execute_input":"2021-07-20T06:01:36.457993Z","iopub.status.idle":"2021-07-20T06:01:42.955237Z","shell.execute_reply.started":"2021-07-20T06:01:36.457957Z","shell.execute_reply":"2021-07-20T06:01:42.954029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport shutil\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.densenet import DenseNet121\n#from keras.applications import inception_v3,mobilenet,vgg19,resnet50,xception # change this line\nfrom keras.applications import inception_v3,mobilenet,vgg19,xception\nfrom tensorflow.python.keras.applications.resnet import ResNet50","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:02:05.479814Z","iopub.execute_input":"2021-07-20T06:02:05.480265Z","iopub.status.idle":"2021-07-20T06:02:05.771177Z","shell.execute_reply.started":"2021-07-20T06:02:05.480219Z","shell.execute_reply":"2021-07-20T06:02:05.770069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reading the total training labels\n#\"train_labels.csv\" will have the labels of all the images\nfile = pd.read_csv(r\"../input/histopathologic-cancer-detection/train_labels.csv\")\n\n#'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2 and 9369c7278ec8bcc6c880d99194de09fc2bd4efbe'\n# these two are the images with full black, which is not needed to train the model\nfile =file[file['id'] != 'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2'] #uncorrect \nfile =file[file['id'] != '9369c7278ec8bcc6c880d99194de09fc2bd4efbe'] #correct\n\nprint(\"total number of images: \",file.shape)\n\n#values_counts will give the total number of different labels\nfile['label'].value_counts()\n\n#since the dataset is biased towards 'label 0'(no tumour) we are taking equal number of data\n#from each label and then concatenating into one variable and shuffle it.\nf_0 = file[file['label'] == 0].sample(80000,random_state = 101)\nf_1 = file[file['label'] == 1].sample(80000,random_state = 101)\nfile = pd.concat([f_0,f_1],axis=0).reset_index(drop = True)\nfile = shuffle(file)\n\nfile['label'].value_counts()\n\n#storing all the labels in variable y\ny = file['label']\n\n#splitting the data for testing and training(20 and 80%) using train_test_split command imported from sklearn\n#random_state will always choose the same data for every trial and stratify is to take equal number of abnormal\n#and normal data from total dataset\nx_train,x_valid = train_test_split(file,test_size = 0.20,random_state= 101,stratify=y)\n\nprint(x_train.shape)\nprint(x_valid.shape)\n\n\nx_train['label'].value_counts()\n\n\nx_valid['label'].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:02:09.380339Z","iopub.execute_input":"2021-07-20T06:02:09.380771Z","iopub.status.idle":"2021-07-20T06:02:10.091553Z","shell.execute_reply.started":"2021-07-20T06:02:09.380735Z","shell.execute_reply":"2021-07-20T06:02:10.090568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#function to create a folder when it doesnt exist\ndef create_folder(folderName):\n    if not os.path.exists(folderName):\n        try:\n            os.makedirs(folderName)\n        except OSError as exc:\n            if exc.errno != errno.EEXIST:\n                raise\n#creating different folders for data/train_dataset and data/valid_dataset\nbase_dir = 'data'\ncreate_folder(base_dir)\n\n\ntrain_dir = os.path.join(base_dir,'train_dataset')\ncreate_folder(train_dir)\nvalid_dir = os.path.join(base_dir,'valid_dataset')\ncreate_folder(valid_dir)\n\n\n#inside train_dataset and valid_dataset folder create two more folders 0 and 1(normal and abnormal)\ntrain_tum = os.path.join(train_dir,'0')\ncreate_folder(train_tum)\ntrain_notum = os.path.join(train_dir,'1')\ncreate_folder(train_notum)\n\nvalid_tum = os.path.join(valid_dir,'0')\ncreate_folder(valid_tum)\nvalid_notum = os.path.join(valid_dir,'1')\ncreate_folder(valid_notum)\n\n\n# check that the folders have been created\nos.listdir('data/train_dataset//')\n\n# Set the id as the index in df_data\nfile.set_index('id', inplace=True)\n\n# Get a list of train and val images\ntrain_list = list(x_train['id'])\nval_list = list(x_valid['id'])","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:02:12.126195Z","iopub.execute_input":"2021-07-20T06:02:12.126593Z","iopub.status.idle":"2021-07-20T06:02:12.19349Z","shell.execute_reply.started":"2021-07-20T06:02:12.126557Z","shell.execute_reply":"2021-07-20T06:02:12.192353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transfer the train images into 0 and 1 respectively\n\nfor image in train_list:\n    \n    # the id in the csv file does not have the .tif extension therefore we add it here\n    fname = image + '.tif'\n    # get the label for a certain image\n    target = file.loc[image,'label']\n    \n    # these must match the folder names\n    if target == 0:\n        label = '0'\n    if target == 1:\n        label = '1'\n    \n    # source path to image\n    src = os.path.join('../input/histopathologic-cancer-detection/train', fname)\n    # destination path to image\n    dst = os.path.join(train_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n\n\n# Transfer the validation images into 0 and 1 respectively\n\nfor image in val_list:\n    \n    # the id in the csv file does not have the .tif extension therefore we add it here\n    fname = image + '.tif'\n    # get the label for a certain image\n    target = file.loc[image,'label']\n    \n    # these must match the folder names\n    if target == 0:\n        label = '0'\n    if target == 1:\n        label = '1'\n    \n\n    # source path to image\n    src = os.path.join('../input/histopathologic-cancer-detection/train', fname)\n    # destination path to image\n    dst = os.path.join(valid_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:02:18.606526Z","iopub.execute_input":"2021-07-20T06:02:18.607161Z","iopub.status.idle":"2021-07-20T06:20:52.448751Z","shell.execute_reply.started":"2021-07-20T06:02:18.607096Z","shell.execute_reply":"2021-07-20T06:20:52.445682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 128\nepochs = 3\n\n#doing data augmentation - (creating more images from availbale images by normalising all the images\n#flipping the images horizontally and vertically)\ndatagen = ImageDataGenerator(rescale=1.0/255,\n                horizontal_flip=True,\n                vertical_flip=True)\n\n#resizing all the images to 96 x 96\ntrain_gen = datagen.flow_from_directory('data/train_dataset/' ,\n                                        target_size = (96,96) ,\n                                        batch_size = batch_size,\n                                       class_mode ='categorical')\n\n\n# def tr_x(tr_gen):\n#     for x,y in tr_gen:\n#         print(x.shape)\n#         yield x\n# def tr_y(tr_gen):\n#     for x,y in tr_gen:\n#         yield y\n\n\nvalid_gen = datagen.flow_from_directory('data/valid_dataset/',target_size = (96,96),batch_size = batch_size, class_mode='categorical')\n\n# def va_x(val_gen):\n#     for x,y in val_gen:\n#         yield x\n# def va_y(val_gen):\n#     for x,y in val_gen:\n#         yield y\n\n#to take only center 32 x 32 patch from the given image\ndef patches(mode):\n    \n    if (mode == 'valid'):\n        xy = valid_gen\n    elif(mode == 'train'):\n        xy = train_gen\n    else:\n        xy = test_gen\n\n    batches = 0\n    for x,y in xy:\n        s = x.shape\n        print(x)\n        img = x[:,32:64,32:64,:]\n        img = np.resize(img,s)\n        batches += 1\n#         yield ([img,y],[y,img])\n        yield img,y\n\npatches('valid')\n","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:20:52.453418Z","iopub.execute_input":"2021-07-20T06:20:52.45384Z","iopub.status.idle":"2021-07-20T06:20:59.866523Z","shell.execute_reply.started":"2021-07-20T06:20:52.453777Z","shell.execute_reply":"2021-07-20T06:20:59.86532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Dropout,Flatten,Dense\n\n#function for building the pretrained architecture\ndef pretrained_model(model):\n    if model == 'densenet':\n        base_model = DenseNet121(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'inception':\n        base_model = inception_v3.InceptionV3(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'mobilenet':\n        base_model = mobilenet.MobileNet(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'vgg':\n        base_model = vgg19.VGG19(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'resnet':\n        base_model = resnet50.ResNet50(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'xception':\n        base_model = xception.Xception(include_top=False,weights='imagenet',input_shape = (96,96,3))\n        \n    for layer in base_model.layers:\n        layer.trainable = False\n        \n    x = base_model.output\n    x = Flatten()(x)\n    x = Dense(150,activation='relu')(x)\n    x = Dropout(0.2)(x)\n    predictions = Dense(2,activation='softmax')(x)\n\n    return tf.keras.models.Model(base_model.input,predictions)   #change this line","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:35:37.555006Z","iopub.execute_input":"2021-07-20T06:35:37.555542Z","iopub.status.idle":"2021-07-20T06:35:37.572669Z","shell.execute_reply.started":"2021-07-20T06:35:37.555494Z","shell.execute_reply":"2021-07-20T06:35:37.571356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Dropout,Flatten,Dense\n\n#function for building the pretrained architecture\ndef pretrained_model(model):\n    if model == 'densenet':\n        base_model = DenseNet121(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'inception':\n        base_model = inception_v3.InceptionV3(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'mobilenet':\n        base_model = mobilenet.MobileNet(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'vgg':\n        base_model = vgg19.VGG19(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'resnet':\n        base_model = resnet50.ResNet50(include_top=False,weights='imagenet',input_shape = (96,96,3))\n    elif model == 'xception':\n        base_model = xception.Xception(include_top=False,weights='imagenet',input_shape = (96,96,3))\n        \n    for layer in base_model.layers:\n        layer.trainable = False\n        \n    x = base_model.output\n    x = Flatten()(x)\n    x = Dense(150,activation='relu')(x)\n    x = Dropout(0.2)(x)\n    predictions = Dense(2,activation='softmax')(x)\n\n    return tf.keras.models.Model(base_model.input,predictions)   #change this line","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:35:40.227872Z","iopub.execute_input":"2021-07-20T06:35:40.228266Z","iopub.status.idle":"2021-07-20T06:35:40.239549Z","shell.execute_reply.started":"2021-07-20T06:35:40.228231Z","shell.execute_reply":"2021-07-20T06:35:40.238385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_model = pretrained_model('vgg')\nprint(main_model.summary())","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:38:33.782887Z","iopub.execute_input":"2021-07-20T06:38:33.78327Z","iopub.status.idle":"2021-07-20T06:38:35.473838Z","shell.execute_reply.started":"2021-07-20T06:38:33.78324Z","shell.execute_reply":"2021-07-20T06:38:35.472629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint,ReduceLROnPlateau,CSVLogger\n#from keras.optimizers import Adam\nfrom tensorflow.keras.optimizers import Adam\n#CSV_Logger is to store all the accuracy and loss values into a csv file for every epoch\n#ModelCheckpoint is to save the best models amoung all the epochs\n#Learning rate starts at 0.001 should keep on reducing at the factor of 0.1 if there is no change in validation accuracy\n\ncsv_logger = CSVLogger(\"result.csv\",separator = \",\",append=True)\n\ncheckpoint_fp = \"vgg_model.h5\"\ncheckpoint = ModelCheckpoint(checkpoint_fp,monitor='val_acc',\n                             verbose=1,save_weights_only=True,\n                            save_best_only= False,mode='max')\n\nlearning_rate = ReduceLROnPlateau(monitor='val_acc',\n                                 factor = 0.1,\n                                 patience = 2,\n                                 verbose = 1,\n                                 mode = 'max',\n                                 min_lr = 0.00001)\n\ncallback = [checkpoint,learning_rate,csv_logger]","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:38:48.071654Z","iopub.execute_input":"2021-07-20T06:38:48.072064Z","iopub.status.idle":"2021-07-20T06:38:48.081435Z","shell.execute_reply.started":"2021-07-20T06:38:48.07203Z","shell.execute_reply":"2021-07-20T06:38:48.079891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps_p_ep_tr =np.ceil(len(x_train)/batch_size)\nsteps_p_ep_va =np.ceil(len(x_valid)/batch_size)\n\n\nmain_model.compile(optimizer = Adam(lr=0.0001),\n              loss = 'binary_crossentropy', metrics=['accuracy'])\n\n#training the model for all the images\nmy_model = main_model.fit_generator(train_gen,\n                                   steps_per_epoch = steps_p_ep_tr,\n                                   validation_data = valid_gen,\n                                   validation_steps = steps_p_ep_va,\n                                   verbose = 1,\n                                   epochs = epochs,\n                                   callbacks = callback)\n\n\n# to remove all the data folder create earlier\nshutil.rmtree('data')\n\n\n# create test_dir\ntest_dir = 'test_dir'\nos.mkdir(test_dir)\n# create test_images inside test_dir\ntest_images = os.path.join(test_dir, 'test_images')\nos.mkdir(test_images)\n\nos.listdir('test_dir/')\n\n\ntest_list = os.listdir('../input/test')\n\n#moving all the test images to test folder\nfor image in test_list:\n    fname = image\n    # source path to image\n    src = os.path.join('../input/test', fname)\n    # destination path to image\n    dst = os.path.join(test_images, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-20T06:39:13.883141Z","iopub.execute_input":"2021-07-20T06:39:13.883484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}