{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport tensorflow as tf\nimport json\nimport seaborn as sns\nfrom PIL import Image\nimport matplotlib.image as mpimg\nimport os\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport random\nfrom imgaug import augmenters as iaa\nimport imgaug as ia\n# import tensorflow.keras\nimport math\nfrom sklearn.utils.class_weight import compute_class_weight\n# from sklearn.model_selection import KFold,StratifiedKFold\n# import keras\n\n# import imblearn.keras\n# from imblearn.keras import balanced_batch_generator\n# from imblearn.over_sampling import RandomOverSampler\n\nDIR = \"../input/cassava-leaf-disease-classification\"\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# kf = KFold(n_splits = 5)\n# skf = StratifiedKFold(n_split = 5,random_state = 7,shuffle=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"with open(\"../input/cassava-leaf-disease-classification/label_num_to_disease_map.json\",'r') as fo:\n    data = json.load(fo)\n# with open(GCS_DS_PATH+\"/label_num_to_disease_map.json\",'r') as fo:\n#     data = json.load(fo)\nfor key,val in data.items():\n    print(key,val)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/cassava-leaf-disease-classification/train.csv\")\ntrain_df.head()\n# print(np.array(train_df['label'].unique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# print(type(train_df['label'].value_counts()))\ncount_df = train_df['label'].value_counts().reset_index()\ncount_df\n# for idx,row in count_df.iterrows():\n#     print(data[str(row['index'])],row['label'])\n    \n# print(count_df['label'].sum())\n# explode = (0, 0.1, 0, 0,0)    \n# fig1, ax1 = plt.subplots()\n# ax1.pie(count_df['label'], \n# #         explode=explode, \n#         labels=count_df['index'], \n#         autopct='%1.1f%%',\n#         shadow=True, startangle=90)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# weight_dict = compute_class_weight(\"balanced\",np.array(train_df['label'].unique()),np.array(train_df['label']))\n# print(weight_dict)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_class_weight(labels_dict,mu=0.15):\n    total = sum(labels_dict.values())\n    keys = labels_dict.keys()\n    class_weight = dict()\n    print(\"total\",total)\n    for key in sorted(keys):\n        print(labels_dict[key])\n#         score = math.log(mu*total/float(labels_dict[key]))\n        score = mu*total/float(labels_dict[key])\n#         class_weight[key] = score if score > 1.0 else 1.0\n        class_weight[int(key)] = score\n\n    return class_weight\n\nlabels_dict = {}\nfor index,row in count_df.iterrows():\n#     print(data[row['index']],row['label'])\n    labels_dict[row['index']] = row['label']\n# labels_dict = {0: 2813, 1: 78, 2: 2814, 3: 78, 4: 7914, 5: 248, 6: 7914, 7: 248}\nclass_weights = create_class_weight(labels_dict)\nprint(class_weights)\n\n# print(sorted(labels_dict))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sometimes = lambda aug: iaa.Sometimes(0.5, aug)\nseq = iaa.Sequential([\n    iaa.Fliplr(0.5), # horizontally flip\n    # sometimes(iaa.AdditiveGaussianNoise(loc=0, scale=(0.0, 0.05), per_channel=0.5)),\n#     sometimes(iaa.Crop(percent=(0, 0.15))),\n#     sometimes(iaa.ElasticTransformation(alpha=(0.5, 3.5), sigma=0.25)),\n    sometimes(iaa.Affine(\n        scale={\"x\": (0.8, 1.2), \"y\": (0.8, 1.2)},\n        translate_percent={\"x\": (-0.2, 0.2), \"y\": (-0.2, 0.2)},\n        rotate=(-10, 10),\n        shear=(-5, 5),\n        order=[0, 1],\n#             cval=(0, 255),\n        mode=ia.ALL\n    )),\n    sometimes(iaa.OneOf([\n#         iaa.ElasticTransformation(alpha=(0.5, 3.5), sigma=0.25),\n        iaa.LinearContrast((0.5, 2.0), per_channel=0.5),\n        iaa.PerspectiveTransform(scale=(0.04, 0.08)),\n#         iaa.Add((-10, 10), per_channel=0.5),\n#         iaa.GaussianBlur((0, 3.0)),\n#         iaa.AverageBlur(k=(2, 7)),\n#         iaa.MedianBlur(k=(3, 11)),\n        ])),\n    iaa.Multiply((0.5, 1.5)),\n    iaa.Sharpen(alpha=(0, 1.0), lightness=(0.75, 1.5)),\n#     iaa.Crop(percent=(0, 0.10)),\n    iaa.Flipud(0.5),\n    iaa.Affine(rotate=(-10, 10), translate_percent={\"x\": (-0.25, 0.25)},shear=(-8, 8), mode='symmetric', cval=(0)),\n#         iaa.PiecewiseAffine(scale=(0.05, 0.1), mode='edge', cval=(0)),\n        \n#     ]),\n], random_order=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BATCH_SIZE=32\nimg_height=300\nimg_width=300\nCOLOR_CHANNEL = 3\nVALIDATION_STEPS= int((len(train_df)*0.12)/BATCH_SIZE)\nNUM_STEPS = int((len(train_df)*0.88)/BATCH_SIZE)\nprint(VALIDATION_STEPS,NUM_STEPS)\nEPOCHS = 100\ntrain_data_length = int((len(train_df)*0.88))\ntraining_folder = '../input/cassava-leaf-disease-classification/train_images/'\ntraining_df = train_df[:train_data_length]\nvalidation_df = train_df[train_data_length:]\n# print(train_data_length)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# def crop_image(img_path,crop_size=300):\n#     img = Image.open(img_path)\n#     img_height,img_width = img.size\n#     img = np.array(img)\n#     x = random.randint(0,img_height-crop_size)\n#     y = random.randint(0,img_width-crop_size)\n    \n#     cropped_img = img[x:x+crop_size,y:y+crop_size,:]\n#     return cropped_img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train_df['label'] = train_df['label'].astype('str')\n# train_iter = tf.keras.preprocessing.image.ImageDataGenerator(validation_split = 0.12,\n#                                      preprocessing_function = None,\n# #                                      featurewise_center = True,\n# #                                      featurewise_std_normalization = True  ,                     \n#                                      rotation_range = 20,\n#                                      zoom_range = 0.15,\n#                                      cval = 0.,\n#                                      width_shift_range=0.2,\n#                                      height_shift_range=0.2,\n#                                      shear_range = 0.15,\n#                                      horizontal_flip = True,\n#                                      vertical_flip = True,\n#                                      fill_mode = 'nearest'\n#                                                             )\n\n\n\n# train_gen = train_iter.flow_from_dataframe(train_df,directory=os.path.join(DIR,'train_images'),subset='training',x_col='image_id',y_col='label',target_size=(img_height,img_width),batch_size=BATCH_SIZE,class_mode='sparse')\n# valid_gen = train_iter.flow_from_dataframe(train_df,directory=os.path.join(DIR,'train_images'),subset='validation',x_col='image_id',y_col='label',target_size=(img_height,img_width),batch_size=BATCH_SIZE,class_mode='sparse')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# x,y = train_gen.next()\n# # print(x[1])\n# for i in range(0,1):\n#     image = x[i]\n# #     print(image)\n#     plt.imshow(image)\n# #     plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# R_mean = []\n# G_mean = []\n# B_mean = []\n\n# R_std = []\n# G_std = []\n# B_std = []\n\n# with tqdm(total=len(train_df)) as pbar:\n#     for idx, row in train_df.iterrows():\n#         pbar.update(1)\n#         img = np.array(Image.open(os.path.join(DIR+\"/train_images/\"+row.image_id)))\n#         R_mean.append(np.mean(img[:,:,0])) \n#         G_mean.append(np.mean(img[:,:,1]))\n#         B_mean.append(np.mean(img[:,:,2]))\n#         R_std.append(np.std(img[:,:,0]))\n#         G_std.append(np.std(img[:,:,1]))\n#         B_std.append(np.std(img[:,:,2]))\n# #         print(np.mean(img[:,:,0]),np.mean(img[:,:,1]),np.mean(img[:,:,2]))\n# RGB_mean = [np.mean(R_mean),np.mean(G_mean),np.mean(B_mean)]\n# RGB_std = [np.std(R_std),np.std(G_std),np.std(B_std)]\n# print(RGB_mean)\n# print(RGB_std)\nRGB_mean = [109.73052931729138, 126.66540463811126, 79.94586978879438]\nRGB_std = [8.8873314840885, 9.080442853826096, 9.415282568048696]\ndef standardise_channel(channel, mean, std):\n    return np.array((channel-mean)/std)\n\ndef standardise_image(image, dataset_mean, dataset_std):\n    standardised_image = np.transpose(np.array([standardise_channel(image[:,:,0], dataset_mean[0], dataset_std[0]),\n                                               standardise_channel(image[:,:,1], dataset_mean[1], dataset_std[1]), \n                                               standardise_channel(image[:,:,2], dataset_mean[2], dataset_std[2])]),\n                                     (1,2,0))\n    return standardised_image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# for idx,row in train_df.iterrows():\n#     img = crop_image(\"../input/cassava-leaf-disease-classification/train_images/\"+row['image_id'])\n#     plt.imshow(img)\n#     break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_model():\n    model = tf.keras.Sequential()\n    effnet = tf.keras.applications.EfficientNetB3(\n    include_top=False, weights=None,input_shape=(img_height,img_width,COLOR_CHANNEL),pooling=None, classifier_activation='softmax'\n    )\n    effnet.load_weights('../input/efficientnetb0b7-keras-weights/efficientnet-b3_weights_tf_dim_ordering_tf_kernels_autoaugment_notop.h5',by_name=True)\n#     effnet.load_weights('../input/efficientnetb0b7-keras-weights/efficientnet-b3_weights_tf_dim_ordering_tf_kernels_autoaugment_notop.h5',by_name=True)\n    model.add(effnet)\n    for layer in model.layers:\n        layer.trainable = True\n    model.add(tf.keras.layers.GlobalAveragePooling2D())\n#     model.add(tf.keras.layers.Dense(512,activation='relu'))\n    model.add(tf.keras.layers.Dense(256,activation='relu'))\n    model.add(tf.keras.layers.BatchNormalization())\n#     model.add(tf.keras.layers.Dropout(0.3))\n    model.add(tf.keras.layers.Dense(5,activation='softmax'))\n\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),loss=tf.keras.losses.sparse_categorical_crossentropy,metrics=[tf.keras.metrics.CategoricalAccuracy()])\n    return model\n\nmodel = create_model()\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def open_and_convert(img_path):\n    img = Image.open(img_path)\n    img_height, img_width = img.size\n    img = img.resize((300,300))\n    img = np.array(img)\n    return img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def custom_generator(dataset,folder, batch_size=BATCH_SIZE, training_mode=True):\n#     print(len(dataset))\n#     batch_val = len(dataset)/32\n#     print(batch_val)\n    while True:\n        for i in range(0,len(dataset),batch_size):\n#             print(dataset['image_id'],dataset['label'])\n#             print(dataset['image_id'])\n#             print(i)\n#             break\n#         break\n#             print(dataset['image_id'][i:batch_size+i])\n#             break\n#         break\n            \n            image_list = dataset['image_id'][i:batch_size+i].to_list()\n            image_list = [folder+image for image in image_list]\n#             print(image_list)\n            \n            label_batch = dataset['label'][i:batch_size+i].to_list()            \n            image_list = [open_and_convert(image) for image in image_list]\n#             num = random.randint(0,31)\n#             print(image_list[num].shape)\n            if training_mode:\n                image_batch = seq.augment_images(images=image_list)\n#                 image_batch = [image/255.0 for image in image_batch]\n            \n            image_batch = [standardise_image(image,RGB_mean,RGB_std) for image in image_list]\n            image_batch = np.array(image_list)\n            label_batch = np.array(label_batch)\n\n            yield image_batch, label_batch","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# cust_gen = custom_generator(validation_df,training_folder, batch_size=BATCH_SIZE, training_mode=True)\n# x,y = next(cust_gen)\n# num = random.randint(0,31)\n# plt.imshow(x[num])\n# # x[num].shape\n# print(type(y))\n# # print(y)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"checkpoint_filepath = './checkpoint8.h5'\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(filepath=checkpoint_filepath, \n                             save_best_only = True, \n                             save_weights_only = True,\n                             monitor = 'val_loss', \n                             mode = 'min', verbose = 1)\nearly_stop = tf.keras.callbacks.EarlyStopping(monitor = 'val_loss', min_delta = 0.001, \n                           patience = 5, mode = 'min', verbose = 1,\n                           restore_best_weights = True)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor = 'val_loss', factor = 0.1, \n                              patience = 2, min_delta = 0.001, \n                              mode = 'min', verbose = 1)\n\nhistory = model.fit_generator(custom_generator(training_df, training_folder, batch_size=BATCH_SIZE, training_mode=True),\n                  steps_per_epoch = NUM_STEPS,\n                  epochs = EPOCHS, \n                  validation_data=custom_generator(validation_df, training_folder, batch_size=BATCH_SIZE,training_mode=True),\n                  validation_steps=VALIDATION_STEPS,\n                  class_weight=class_weights,\n                  callbacks=[model_checkpoint_callback,early_stop, reduce_lr])\n\n# model.fit_generator(\n#     train_gen,  \n#     steps_per_epoch = NUM_STEPS,\n#     epochs = EPOCHS,\n#     validation_data = valid_gen, \n#     validation_steps = VALIDATION_STEPS,\n#     class_weight = class_weights,\n#     callbacks = [model_checkpoint_callback,early_stop, reduce_lr]) \n\nmodel.save('./cassava8.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_img = Image.open(os.path.join(DIR,'test_images/2216849948.jpg'))\nprint(test_img.size)  \ntest_img = test_img.resize((img_height, img_width))\nprint(test_img.size)\ntest_img = np.expand_dims(test_img, axis = 0)   \nprint(test_img.shape)\nmodel = tf.keras.models.load_model(\"./cassava8.h5\")\nprint(model.predict(test_img))\nprint(np.argmax(model.predict(test_img))) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df = pd.read_csv(\"../input/cassava-leaf-disease-classification/sample_submission.csv\")\nsubmission_df.head()\npredicted_classes = []\nfor img_id in submission_df['image_id']:\n    test_img = Image.open(os.path.join(DIR,'test_images/'+img_id))\n    test_img = test_img.resize((img_height, img_width))\n    test_img = np.expand_dims(test_img, axis = 0)\n    predicted_classes.append(np.argmax(model.predict(test_img)))\n    \nsubmission_df['label'] = predicted_classes\nsubmission_df.head()\n    \n\nsubmission_df.to_csv(\"submission.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# model.load(\"./cassava.h5\")\n# model.predict(valid_gen, batch_size=BATCH_SIZE, verbose=1, steps=VALIDATION_STEPS)\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}