{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.utils import to_categorical, Sequence\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization\nfrom keras.optimizers import RMSprop,Adam\nfrom keras.applications import ResNet50, ResNet101, DenseNet121","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = '/kaggle/input/cassava-leaf-disease-classification/'\nos.listdir(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_bar(data, name):\n    data_label = data[name].value_counts().sort_index()\n    dict_train = dict(zip(data_label.keys(), ((data_label.sort_index())).tolist()))\n    names = list(dict_train.keys())\n    values = list(dict_train.values())\n    plt.bar(names, values)\n    plt.grid()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.read_csv(path+'train.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('number of train data:', len(train_data))\nprint('number of train images:', len(os.listdir(path+'train_images/')))\nprint('number of test images:', len(os.listdir(path+'test_images/')))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_bar(train_data, 'label')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = cv2.imread(path+'train_images/'+'1000015157.jpg')\nplt.imshow(img)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 3\nimg_size = 512\nimg_channel = 3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train = to_categorical(train_data['label'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_weight = dict(zip(range(0, 5), (train_data['label'].value_counts().sort_index()/len(train_data))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class DataGenerator(Sequence):\n    def __init__(self, path, list_IDs, labels, batch_size, img_size, img_channel):\n        self.path = path\n        self.list_IDs = list_IDs\n        self.labels = labels\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_channel = img_channel\n        self.indexes = np.arange(len(self.list_IDs))\n        \n    def __len__(self):\n        return int(np.floor(len(self.list_IDs)/self.batch_size))\n    \n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        X, y = self.__data_generation(list_IDs_temp)\n        return X, y\n\n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.empty((self.batch_size, self.img_size, self.img_size, self.img_channel))\n        y = np.empty((self.batch_size, 5), dtype=int)\n        for i, ID in enumerate(list_IDs_temp):\n            data_file = cv2.imread(self.path+ID)\n            img = cv2.resize(data_file, (self.img_size, self.img_size))\n            X[i, ] = img\n            y[i, ] = self.labels[i]\n        X = X.astype('float32')\n        X -= X.mean()\n        X /= X.std()\n        return X, y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"resnet_weights_path = '../input/resnet50/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"conv_base = ResNet50(weights=None,\n                     include_top=True,\n                     input_shape=(img_size, img_size, img_channel))\n#conv_base = DenseNet121(weights='../input/models/densenet121_weights_tf_dim_ordering_tf_kernels.h5',\n#                     include_top=True,\n#                     input_shape=(img_size, img_size, img_channel))\nconv_base.trainable = True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\n#model.add(Conv2D(128, input_shape=(img_size,img_size,img_channel), kernel_size=5, strides=4, activation='relu'))\n#model.add(BatchNormalization())\n#model.add(MaxPool2D(pool_size=(2)))\n#model.add(Conv2D(128, kernel_size=5, activation='relu'))\n#model.add(BatchNormalization())\n#model.add(MaxPool2D(pool_size=(2)))\n#model.add(Conv2D(256, kernel_size=5, activation='relu'))\n#model.add(BatchNormalization())\n#model.add(MaxPool2D(pool_size=(2)))\nmodel.add(conv_base)\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(5, activation='softmax'))\n#model.add(Dense(5, activation='sigmoid'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# coding: utf8\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import models\n\n\n#\n# image dimensions\n#\n\nimg_height = 512\nimg_width = 512\nimg_channels = 3\n\n#\n# network params\n#\n\ncardinality = 32\n\n\ndef residual_network(x):\n    \"\"\"\n    ResNeXt by default. For ResNet set `cardinality` = 1 above.\n    \n    \"\"\"\n    def add_common_layers(y):\n        y = layers.BatchNormalization()(y)\n        y = layers.LeakyReLU()(y)\n\n        return y\n\n    def grouped_convolution(y, nb_channels, _strides):\n        # when `cardinality` == 1 this is just a standard convolution\n        if cardinality == 1:\n            return layers.Conv2D(nb_channels, kernel_size=(3, 3), strides=_strides, padding='same')(y)\n        \n        assert not nb_channels % cardinality\n        _d = nb_channels // cardinality\n\n        # in a grouped convolution layer, input and output channels are divided into `cardinality` groups,\n        # and convolutions are separately performed within each group\n        groups = []\n        for j in range(cardinality):\n            group = layers.Lambda(lambda z: z[:, :, :, j * _d:j * _d + _d])(y)\n            groups.append(layers.Conv2D(_d, kernel_size=(3, 3), strides=_strides, padding='same')(group))\n            \n        # the grouped convolutional layer concatenates them as the outputs of the layer\n        y = layers.concatenate(groups)\n\n        return y\n\n    def residual_block(y, nb_channels_in, nb_channels_out, _strides=(1, 1), _project_shortcut=False):\n        \"\"\"\n        Our network consists of a stack of residual blocks. These blocks have the same topology,\n        and are subject to two simple rules:\n\n        - If producing spatial maps of the same size, the blocks share the same hyper-parameters (width and filter sizes).\n        - Each time the spatial map is down-sampled by a factor of 2, the width of the blocks is multiplied by a factor of 2.\n        \"\"\"\n        shortcut = y\n\n        # we modify the residual building block as a bottleneck design to make the network more economical\n        y = layers.Conv2D(nb_channels_in, kernel_size=(1, 1), strides=(1, 1), padding='same')(y)\n        y = add_common_layers(y)\n\n        # ResNeXt (identical to ResNet when `cardinality` == 1)\n        y = grouped_convolution(y, nb_channels_in, _strides=_strides)\n        y = add_common_layers(y)\n\n        y = layers.Conv2D(nb_channels_out, kernel_size=(1, 1), strides=(1, 1), padding='same')(y)\n        # batch normalization is employed after aggregating the transformations and before adding to the shortcut\n        y = layers.BatchNormalization()(y)\n\n        # identity shortcuts used directly when the input and output are of the same dimensions\n        if _project_shortcut or _strides != (1, 1):\n            # when the dimensions increase projection shortcut is used to match dimensions (done by 1×1 convolutions)\n            # when the shortcuts go across feature maps of two sizes, they are performed with a stride of 2\n            shortcut = layers.Conv2D(nb_channels_out, kernel_size=(1, 1), strides=_strides, padding='same')(shortcut)\n            shortcut = layers.BatchNormalization()(shortcut)\n\n        y = layers.add([shortcut, y])\n\n        # relu is performed right after each batch normalization,\n        # expect for the output of the block where relu is performed after the adding to the shortcut\n        y = layers.LeakyReLU()(y)\n\n        return y\n\n    # conv1\n    x = layers.Conv2D(256, kernel_size=(5, 5), strides=(2, 2), padding='same')(x)\n    x = add_common_layers(x)\n\n    # conv2\n    x = layers.MaxPool2D(pool_size=(3, 3), strides=(2, 2), padding='same')(x)\n    for i in range(3):#3\n        project_shortcut = True if i == 0 else False\n        x = residual_block(x, 64, 128, _project_shortcut=project_shortcut)\n\n    # conv3\n    for i in range(4):#4\n        # down-sampling is performed by conv3_1, conv4_1, and conv5_1 with a stride of 2\n        strides = (2, 2) if i == 0 else (1, 1)\n        x = residual_block(x, 128, 256, _strides=strides)\n\n    # conv4\n    for i in range(6):#6\n        strides = (2, 2) if i == 0 else (1, 1)\n        x = residual_block(x, 256, 512, _strides=strides)\n\n    # conv5\n    for i in range(3):#3\n        strides = (2, 2) if i == 0 else (1, 1)\n        x = residual_block(x, 512, 2048, _strides=strides)\n\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dense(5,activation='softmax')(x)\n\n    return x\n\n\nimage_tensor = layers.Input(shape=(img_height, img_width, img_channels))\nnetwork_output = residual_network(image_tensor)\n  \nmodel = models.Model(inputs=[image_tensor], outputs=[network_output])\nprint(model.summary())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(optimizer=Adam(lr=1e-3), loss='categorical_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model Training"},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 100","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_generator = DataGenerator(path+'train_images/', train_data['image_id'], y_train, batch_size, img_size, img_channel)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              epochs = epochs,\n                              class_weight = None,\n                              workers=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_generator = DataGenerator(path+'test_images/', samp_subm['image_id'], samp_subm['label'], 1, img_size, img_channel)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predict = model.predict_generator(test_generator, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samp_subm['label'] = predict.argmax(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samp_subm.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}