{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## First step is to install the very good ImageDataAugmentor Library from mjkvaak under MIT licence.","metadata":{}},{"cell_type":"code","source":"pip install git+https://github.com/mjkvaak/ImageDataAugmentor","metadata":{"execution":{"iopub.status.busy":"2022-04-14T16:55:23.786106Z","iopub.execute_input":"2022-04-14T16:55:23.786626Z","iopub.status.idle":"2022-04-14T16:55:37.358171Z","shell.execute_reply.started":"2022-04-14T16:55:23.786523Z","shell.execute_reply":"2022-04-14T16:55:37.357110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Attempting to predict Sorghum species with pretrained model :\n\n* We will use the given dataset of 22194 1024x1024 images\n\n* We will use data augmentation via Flow_from_dataframe resizing the image 260x260\n\n* We will try EfficientNet B1 in this notebook","metadata":{}},{"cell_type":"markdown","source":"# Importing my favorite libraries for Deep learning activities","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport sys\nimport os\nimport math\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport PIL\nimport tensorflow as tf\nimport random\nfrom IPython.core.debugger import set_trace\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import Sequential, layers\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.data import Dataset\nimport cv2\nimport tensorflow_addons as tfa\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.callbacks import LearningRateScheduler\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nimport albumentations as A\nfrom albumentations.core.composition import Compose, OneOf\nimport datetime\nfrom ImageDataAugmentor.image_data_augmentor import *","metadata":{"papermill":{"duration":5.788397,"end_time":"2022-03-30T07:46:56.84275","exception":false,"start_time":"2022-03-30T07:46:51.054353","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:05:57.356613Z","iopub.execute_input":"2022-04-14T17:05:57.357156Z","iopub.status.idle":"2022-04-14T17:06:06.077318Z","shell.execute_reply.started":"2022-04-14T17:05:57.357099Z","shell.execute_reply":"2022-04-14T17:06:06.076344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We define the PATH and Batch_size :","metadata":{}},{"cell_type":"code","source":"# Load the TensorBoard notebook extension\n%load_ext tensorboard","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:06.079150Z","iopub.execute_input":"2022-04-14T17:06:06.079388Z","iopub.status.idle":"2022-04-14T17:06:06.089433Z","shell.execute_reply.started":"2022-04-14T17:06:06.079359Z","shell.execute_reply":"2022-04-14T17:06:06.088351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/sorghum-id-fgvc-9/'\ntrain_path = PATH+'train_images/'\ntest_path = PATH+'test/'\nsave_path = '/kaggle/working/'\nbatch_size = 12\nepoch = 100\nWIDTH = 512\nHEIGHT = 512","metadata":{"papermill":{"duration":0.045996,"end_time":"2022-03-30T07:46:56.929131","exception":false,"start_time":"2022-03-30T07:46:56.883135","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:06.091191Z","iopub.execute_input":"2022-04-14T17:06:06.092162Z","iopub.status.idle":"2022-04-14T17:06:06.107052Z","shell.execute_reply.started":"2022-04-14T17:06:06.092125Z","shell.execute_reply":"2022-04-14T17:06:06.106075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create our Dataframe for training :","metadata":{}},{"cell_type":"code","source":"image_df = pd.read_csv(PATH+'train_cultivar_mapping.csv')","metadata":{"papermill":{"duration":0.075762,"end_time":"2022-03-30T07:46:57.305354","exception":false,"start_time":"2022-03-30T07:46:57.229592","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:10.929858Z","iopub.execute_input":"2022-04-14T17:06:10.930447Z","iopub.status.idle":"2022-04-14T17:06:10.978875Z","shell.execute_reply.started":"2022-04-14T17:06:10.930373Z","shell.execute_reply":"2022-04-14T17:06:10.978090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df.dropna(inplace=True)","metadata":{"papermill":{"duration":0.086915,"end_time":"2022-03-30T07:46:57.620511","exception":false,"start_time":"2022-03-30T07:46:57.533596","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:11.105434Z","iopub.execute_input":"2022-04-14T17:06:11.105909Z","iopub.status.idle":"2022-04-14T17:06:11.120973Z","shell.execute_reply.started":"2022-04-14T17:06:11.105876Z","shell.execute_reply":"2022-04-14T17:06:11.119949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df","metadata":{"papermill":{"duration":0.104566,"end_time":"2022-03-30T07:46:57.465979","exception":false,"start_time":"2022-03-30T07:46:57.361413","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:11.311597Z","iopub.execute_input":"2022-04-14T17:06:11.311922Z","iopub.status.idle":"2022-04-14T17:06:11.335158Z","shell.execute_reply.started":"2022-04-14T17:06:11.311882Z","shell.execute_reply":"2022-04-14T17:06:11.334204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Debugging the notebook with lower values \n\n# epoch = 20\n# image_df = image_df[:200]\n# image_df","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:11.503842Z","iopub.execute_input":"2022-04-14T17:06:11.504820Z","iopub.status.idle":"2022-04-14T17:06:11.508952Z","shell.execute_reply.started":"2022-04-14T17:06:11.504756Z","shell.execute_reply":"2022-04-14T17:06:11.508296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kfold = StratifiedKFold(n_splits=4, shuffle=True)\n\nfor train_index, valid_index in kfold.split(image_df['image'],image_df['cultivar']):\n    train_images, valid_images = image_df['image'].iloc[train_index], image_df['image'].iloc[valid_index]\n    train_cultivar, valid_cultivar = image_df['cultivar'].iloc[train_index], image_df['cultivar'].iloc[valid_index]","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:11.667265Z","iopub.execute_input":"2022-04-14T17:06:11.668269Z","iopub.status.idle":"2022-04-14T17:06:11.743492Z","shell.execute_reply.started":"2022-04-14T17:06:11.668205Z","shell.execute_reply":"2022-04-14T17:06:11.742334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df= pd.DataFrame({'image':train_images, 'cultivar':train_cultivar})\nval_df= pd.DataFrame({'image':valid_images, 'cultivar':valid_cultivar})","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:11.851739Z","iopub.execute_input":"2022-04-14T17:06:11.852055Z","iopub.status.idle":"2022-04-14T17:06:11.859934Z","shell.execute_reply.started":"2022-04-14T17:06:11.852021Z","shell.execute_reply":"2022-04-14T17:06:11.858606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df), len(val_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:12.019750Z","iopub.execute_input":"2022-04-14T17:06:12.020090Z","iopub.status.idle":"2022-04-14T17:06:12.027025Z","shell.execute_reply.started":"2022-04-14T17:06:12.020053Z","shell.execute_reply":"2022-04-14T17:06:12.025961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df, val_df = train_test_split(image_df, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:12.209400Z","iopub.execute_input":"2022-04-14T17:06:12.209865Z","iopub.status.idle":"2022-04-14T17:06:12.213565Z","shell.execute_reply.started":"2022-04-14T17:06:12.209832Z","shell.execute_reply":"2022-04-14T17:06:12.212631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Images are 1024x1024 RGB in a .png format\n\n# Let's create our Train and Validation datasets (pair of tensors with images preprocessed and targets) :","metadata":{"papermill":{"duration":0.066834,"end_time":"2022-03-30T07:46:57.754387","exception":false,"start_time":"2022-03-30T07:46:57.687553","status":"completed"},"tags":[]}},{"cell_type":"code","source":"transform = Compose([\n            A.RandomResizedCrop(height=HEIGHT, width=WIDTH),\n            A.Flip(p=0.5),\n            A.RandomRotate90(p=0.5),\n            A.ShiftScaleRotate(p=0.5),\n            A.HueSaturationValue(p=0.5),\n            A.OneOf([\n                A.RandomBrightnessContrast(p=0.5),\n                A.RandomGamma(p=0.5),\n            ], p=0.5),\n            A.OneOf([\n                A.Blur(p=0.1),\n                A.GaussianBlur(p=0.1),\n                A.MotionBlur(p=0.1),\n            ], p=0.1),\n            A.OneOf([\n                A.GaussNoise(p=0.1),\n                A.ISONoise(p=0.1),\n                A.GridDropout(ratio=0.5, p=0.2),\n                A.CoarseDropout(max_holes=16, min_holes=8, max_height=16, max_width=16, min_height=8, min_width=8, p=0.2)\n            ], p=0.2)\n        ])","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:12.528187Z","iopub.execute_input":"2022-04-14T17:06:12.529370Z","iopub.status.idle":"2022-04-14T17:06:12.538888Z","shell.execute_reply.started":"2022-04-14T17:06:12.529325Z","shell.execute_reply":"2022-04-14T17:06:12.537801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataAugmentor(augment=transform)\nval_datagen = ImageDataAugmentor()\n","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:12.754003Z","iopub.execute_input":"2022-04-14T17:06:12.754953Z","iopub.status.idle":"2022-04-14T17:06:12.760260Z","shell.execute_reply.started":"2022-04-14T17:06:12.754901Z","shell.execute_reply":"2022-04-14T17:06:12.759534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_augmented = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    shuffle=True,\n    directory=train_path,\n    x_col='image',\n    y_col='cultivar',\n    class_mode='categorical',\n    target_size=(HEIGHT,WIDTH),\n    batch_size=batch_size)","metadata":{"papermill":{"duration":34.97662,"end_time":"2022-03-30T07:47:32.936606","exception":false,"start_time":"2022-03-30T07:46:57.959986","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:12.987743Z","iopub.execute_input":"2022-04-14T17:06:12.988510Z","iopub.status.idle":"2022-04-14T17:06:26.056756Z","shell.execute_reply.started":"2022-04-14T17:06:12.988460Z","shell.execute_reply":"2022-04-14T17:06:26.055817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_augmented = val_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    shuffle=True,\n    directory=train_path,\n    x_col='image',\n    y_col='cultivar',\n    class_mode='categorical',\n    target_size=(HEIGHT,WIDTH),\n    batch_size=batch_size)","metadata":{"papermill":{"duration":7.957408,"end_time":"2022-03-30T07:47:40.935384","exception":false,"start_time":"2022-03-30T07:47:32.977976","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:26.058448Z","iopub.execute_input":"2022-04-14T17:06:26.058705Z","iopub.status.idle":"2022-04-14T17:06:30.629850Z","shell.execute_reply.started":"2022-04-14T17:06:26.058670Z","shell.execute_reply":"2022-04-14T17:06:30.628720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_classes = len(train_augmented.class_indices)\nnum_classes","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:30.631354Z","iopub.execute_input":"2022-04-14T17:06:30.631721Z","iopub.status.idle":"2022-04-14T17:06:30.638149Z","shell.execute_reply.started":"2022-04-14T17:06:30.631674Z","shell.execute_reply":"2022-04-14T17:06:30.637432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_id, num_images = np.unique(train_augmented.classes,return_counts=True)\nmax_value = max(num_images)\nclass_weights = {c : max_value/n for c,n in zip(class_id, num_images)}","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:30.640194Z","iopub.execute_input":"2022-04-14T17:06:30.640478Z","iopub.status.idle":"2022-04-14T17:06:30.653297Z","shell.execute_reply.started":"2022-04-14T17:06:30.640436Z","shell.execute_reply":"2022-04-14T17:06:30.652555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's plot 9 images to see if is works :","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nbatch=train_augmented.next()\nfor i in range(9):\n    ax = plt.subplot(3, 3, i + 1)\n    plt.imshow(batch[0][i].astype(np.uint8))\n    plt.title(batch[0][i].shape)\n    plt.axis(\"off\")","metadata":{"papermill":{"duration":6.538954,"end_time":"2022-03-30T07:47:47.51561","exception":false,"start_time":"2022-03-30T07:47:40.976656","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:30.656679Z","iopub.execute_input":"2022-04-14T17:06:30.657119Z","iopub.status.idle":"2022-04-14T17:06:32.850754Z","shell.execute_reply.started":"2022-04-14T17:06:30.657082Z","shell.execute_reply":"2022-04-14T17:06:32.849728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nbatch=val_augmented.next()\nfor i in range(9):\n    ax = plt.subplot(3, 3, i + 1)\n    plt.imshow(batch[0][i].astype(np.uint8))\n    plt.title(batch[0][i].shape)\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:32.852527Z","iopub.execute_input":"2022-04-14T17:06:32.852795Z","iopub.status.idle":"2022-04-14T17:06:34.488523Z","shell.execute_reply.started":"2022-04-14T17:06:32.852762Z","shell.execute_reply":"2022-04-14T17:06:34.487568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# It's time to prepare our model and start the training :","metadata":{}},{"cell_type":"code","source":"def plot_history(history, title='', axs=None, exp_name=\"\"): # This is the simple function to plot our training and validation curves\n    if axs is not None:\n        ax1, ax2 = axs\n    else:\n        f, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 4))\n    \n    if len(exp_name) > 0 and exp_name[0] != '_':\n        exp_name = '_' + exp_name\n    ax1.plot(history.history['loss'], label='train' + exp_name)\n    ax1.plot(history.history['val_loss'], label='val' + exp_name)\n    ax1.set_ylim(0., 4)\n    ax1.set_title('loss')\n    ax1.legend()\n\n    ax2.plot(history.history['accuracy'], label='train accuracy'  + exp_name)\n    ax2.plot(history.history['val_accuracy'], label='val_accuracy'  + exp_name)\n    ax2.set_ylim(0, 1)\n    ax2.set_title('Accuracy')\n    ax2.legend()\n    return (ax1, ax2)","metadata":{"papermill":{"duration":0.066367,"end_time":"2022-03-30T07:47:47.63634","exception":false,"start_time":"2022-03-30T07:47:47.569973","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:34.489832Z","iopub.execute_input":"2022-04-14T17:06:34.490078Z","iopub.status.idle":"2022-04-14T17:06:34.499943Z","shell.execute_reply.started":"2022-04-14T17:06:34.490047Z","shell.execute_reply":"2022-04-14T17:06:34.498979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_model(): # Here we choose our model : Efficientnet B1 pretrained with ImageNet dataset with an input shape of 260x260\n    model = tf.keras.applications.EfficientNetB0(include_top=False,weights='imagenet',input_shape=(HEIGHT,WIDTH,3))\n    return model","metadata":{"papermill":{"duration":6.117054,"end_time":"2022-03-30T07:47:53.807006","exception":false,"start_time":"2022-03-30T07:47:47.689952","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:34.501280Z","iopub.execute_input":"2022-04-14T17:06:34.501927Z","iopub.status.idle":"2022-04-14T17:06:34.512862Z","shell.execute_reply.started":"2022-04-14T17:06:34.501889Z","shell.execute_reply":"2022-04-14T17:06:34.512114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_nontrainable_layers(model): # We define trainability for the base model\n    model.trainable=False\n    return model\n\ndef set_trainable_layers(model): \n    model.trainable=True\n    return model\n\ndef set_quartertrainable_layers(model):\n    model.trainable=True\n    for layer in model.layers:\n        layer.trainable = False\n    for layer in model.layers[-20:]:\n        if not isinstance(layer, layers.BatchNormalization):\n            layer.trainable = True\n    return model\n    \ndef set_halftrainable_layers(model):\n    model.trainable=True\n    for layer in model.layers[-round(len(model.layers)*0.5):]:\n        if not isinstance(layer, layers.BatchNormalization):\n            layer.trainable = True\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:34.514121Z","iopub.execute_input":"2022-04-14T17:06:34.514538Z","iopub.status.idle":"2022-04-14T17:06:34.526525Z","shell.execute_reply.started":"2022-04-14T17:06:34.514492Z","shell.execute_reply":"2022-04-14T17:06:34.525591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_last_layers(model): # Here we complete our model with last layers for our problem at hand\n    input_layer = tf.keras.Input(shape=(HEIGHT,WIDTH,3))\n#     base_model = set_quartertrainable_layers(model)\n#     base_model = set_halftrainable_layers(model)\n#     base_model = set_nontrainable_layers(model)\n    base_model = set_trainable_layers(model)\n    flatten_layer = layers.Flatten()\n    global_layer = layers.GlobalAveragePooling2D()\n    dense_layer = layers.Dense(256, activation='relu', kernel_initializer='he_uniform')\n    dropout_layer = layers.Dropout(0.5)\n    prediction_layer = layers.Dense(num_classes, activation='softmax')\n    \n    model = Sequential([\n        input_layer,\n        base_model,\n        global_layer,\n        dropout_layer,\n#         flatten_layer,\n#         dense_layer,\n        prediction_layer\n    ])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:34.528887Z","iopub.execute_input":"2022-04-14T17:06:34.529166Z","iopub.status.idle":"2022-04-14T17:06:34.540310Z","shell.execute_reply.started":"2022-04-14T17:06:34.529123Z","shell.execute_reply":"2022-04-14T17:06:34.539311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(): # We assemble our model and compile it with the proper loss function and metrics for image classification\n    model = load_model()\n    model = add_last_layers(model)\n    \n    opt = tf.keras.optimizers.Adam(learning_rate=0.01)\n    model.compile(loss='categorical_crossentropy',\n                 optimizer=opt,\n                 metrics=['accuracy'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:34.541910Z","iopub.execute_input":"2022-04-14T17:06:34.542160Z","iopub.status.idle":"2022-04-14T17:06:34.557042Z","shell.execute_reply.started":"2022-04-14T17:06:34.542131Z","shell.execute_reply":"2022-04-14T17:06:34.556005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_effnet= build_model()\nmodel_effnet.summary()","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:34.558533Z","iopub.execute_input":"2022-04-14T17:06:34.558785Z","iopub.status.idle":"2022-04-14T17:06:37.349506Z","shell.execute_reply.started":"2022-04-14T17:06:34.558754Z","shell.execute_reply":"2022-04-14T17:06:37.348215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_accuracy',\n                   patience=7,\n                   verbose=1,\n                   restore_best_weights=True)\n\ncp = ModelCheckpoint(save_path + 'effnetB1.ckpt',\n                     monitor='val_loss',\n                     verbose=1,\n                     save_best_only=True,\n                     save_weights_only=False,\n                     mode='min' )\n\ncsv = tf.keras.callbacks.CSVLogger('history.csv')\n\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', \n                                                 factor=0.4,\n                                                 verbose=1,\n                                                 patience=2, \n                                                 min_lr=0.00001)","metadata":{"papermill":{"duration":0.066896,"end_time":"2022-03-30T07:47:55.200302","exception":false,"start_time":"2022-03-30T07:47:55.133406","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:37.351029Z","iopub.execute_input":"2022-04-14T17:06:37.351351Z","iopub.status.idle":"2022-04-14T17:06:37.359070Z","shell.execute_reply.started":"2022-04-14T17:06:37.351301Z","shell.execute_reply":"2022-04-14T17:06:37.358162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_augmented.n//train_augmented.batch_size\nSTEP_SIZE_VALID = val_augmented.n//val_augmented.batch_size","metadata":{"execution":{"iopub.status.busy":"2022-04-14T17:06:37.360259Z","iopub.execute_input":"2022-04-14T17:06:37.360643Z","iopub.status.idle":"2022-04-14T17:06:37.377889Z","shell.execute_reply.started":"2022-04-14T17:06:37.360606Z","shell.execute_reply":"2022-04-14T17:06:37.376817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nhistory = model_effnet.fit(train_augmented,\n                    epochs=epoch,\n                    steps_per_epoch=STEP_SIZE_TRAIN,\n                    callbacks=[es,cp,reduce_lr,csv],\n                    verbose=1,\n                    class_weight=class_weights,\n                    validation_data=val_augmented,\n                    validation_steps=STEP_SIZE_VALID)","metadata":{"papermill":{"duration":9367.330539,"end_time":"2022-03-30T10:24:02.589412","exception":false,"start_time":"2022-03-30T07:47:55.258873","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T17:06:37.379543Z","iopub.execute_input":"2022-04-14T17:06:37.379808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_history(history)","metadata":{"papermill":{"duration":11.347117,"end_time":"2022-03-30T10:24:17.179815","exception":false,"start_time":"2022-03-30T10:24:05.832698","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now we can use our trained model to predict the labels for test images :","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv(PATH+'sample_submission.csv')\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.462557Z","iopub.status.idle":"2022-04-14T12:19:35.463235Z","shell.execute_reply.started":"2022-04-14T12:19:35.463042Z","shell.execute_reply":"2022-04-14T12:19:35.463072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen= ImageDataAugmentor()","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.464222Z","iopub.status.idle":"2022-04-14T12:19:35.464675Z","shell.execute_reply.started":"2022-04-14T12:19:35.4645Z","shell.execute_reply":"2022-04-14T12:19:35.464518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = test_gen.flow_from_dataframe(dataframe=submission,\n                                              directory=test_path,\n                                              x_col='filename',\n                                              y_col=None,\n                                              target_size=(WIDTH,HEIGHT),\n                                              color_mode='rgb',\n                                              class_mode=None,\n                                              batch_size=1,\n                                              shuffle=False,)","metadata":{"papermill":{"duration":0.760723,"end_time":"2022-03-30T10:24:18.696646","exception":false,"start_time":"2022-03-30T10:24:17.935923","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-14T12:19:35.465571Z","iopub.status.idle":"2022-04-14T12:19:35.46601Z","shell.execute_reply.started":"2022-04-14T12:19:35.465841Z","shell.execute_reply":"2022-04-14T12:19:35.465858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TEST=test_generator.n//test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.466924Z","iopub.status.idle":"2022-04-14T12:19:35.467257Z","shell.execute_reply.started":"2022-04-14T12:19:35.467078Z","shell.execute_reply":"2022-04-14T12:19:35.467102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TEST,test_generator.n,test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.46837Z","iopub.status.idle":"2022-04-14T12:19:35.468718Z","shell.execute_reply.started":"2022-04-14T12:19:35.468538Z","shell.execute_reply":"2022-04-14T12:19:35.46857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\ntest_generator.reset()\nresults = model_effnet.predict(test_generator,verbose=1,steps=STEP_SIZE_TEST)","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.469562Z","iopub.status.idle":"2022-04-14T12:19:35.469885Z","shell.execute_reply.started":"2022-04-14T12:19:35.469711Z","shell.execute_reply":"2022-04-14T12:19:35.469734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_results = np.argmax(results,axis=1)\n# label_dict = pd.read_csv(PATH+'train_cultivar_mapping.csv').drop(columns=['image'])\n# best_cultivar = []\n# for result in best_results:\n#     best_cultivar.append(label_dict.iloc[result].values[0])\n# submission['cultivar']=best_cultivar\n","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.471144Z","iopub.status.idle":"2022-04-14T12:19:35.471513Z","shell.execute_reply.started":"2022-04-14T12:19:35.471305Z","shell.execute_reply":"2022-04-14T12:19:35.471327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_class_indices=np.argmax(results,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.472328Z","iopub.status.idle":"2022-04-14T12:19:35.472693Z","shell.execute_reply.started":"2022-04-14T12:19:35.472509Z","shell.execute_reply":"2022-04-14T12:19:35.472532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = (train_augmented.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\npredictions = [labels[k] for k in predicted_class_indices]","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.473803Z","iopub.status.idle":"2022-04-14T12:19:35.474139Z","shell.execute_reply.started":"2022-04-14T12:19:35.47395Z","shell.execute_reply":"2022-04-14T12:19:35.473973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames=test_generator.filenames\n","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.475904Z","iopub.status.idle":"2022-04-14T12:19:35.476222Z","shell.execute_reply.started":"2022-04-14T12:19:35.476052Z","shell.execute_reply":"2022-04-14T12:19:35.476075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.DataFrame({\"Filename\":[filename.replace('all_classes/','')for filename in filenames],\n                      \"cultivar\":predictions})","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.47768Z","iopub.status.idle":"2022-04-14T12:19:35.47823Z","shell.execute_reply.started":"2022-04-14T12:19:35.47805Z","shell.execute_reply":"2022-04-14T12:19:35.478071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.479662Z","iopub.status.idle":"2022-04-14T12:19:35.47998Z","shell.execute_reply.started":"2022-04-14T12:19:35.479811Z","shell.execute_reply":"2022-04-14T12:19:35.479834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# And finally we create our submission.csv in order to scor our prediction :","metadata":{}},{"cell_type":"code","source":"submission.to_csv('submission_effnet_dataaug.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.481006Z","iopub.status.idle":"2022-04-14T12:19:35.481657Z","shell.execute_reply.started":"2022-04-14T12:19:35.481456Z","shell.execute_reply":"2022-04-14T12:19:35.481478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## For the moment, the test accuracy remains very low, I am grateful for any comment or suggestion on how to achieve better accuracy with my code.\n\n# Have a nice day","metadata":{}},{"cell_type":"code","source":"!kaggle competitions submit -c sorghum-id-fgvc-9 -f submission_effnet_dataaug.csv -m \"With flow from dataframe generator\"","metadata":{"execution":{"iopub.status.busy":"2022-04-14T12:19:35.48268Z","iopub.status.idle":"2022-04-14T12:19:35.482989Z","shell.execute_reply.started":"2022-04-14T12:19:35.482825Z","shell.execute_reply":"2022-04-14T12:19:35.482846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}