{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n!pip install opencv-contrib-python\nimport cv2\nprint(os.listdir(\"/kaggle/input\"))\nimport matplotlib.pyplot as plt\n%matplotlib inline \nimport json\nimport glob\nimport random\nfrom IPython.display import display, display_markdown\nfrom math import floor\nimport matplotlib.image as mpimg\nimport matplotlib.patches as patches\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nimport seaborn as sns\nfrom PIL import Image, ImageFile\nfrom tqdm.notebook import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models\nfrom sklearn.model_selection import train_test_split\nfrom torchvision import transforms\nfrom skimage import io\nImageFile.LOAD_TRUNCATED_IMAGES = True","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade pip","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training Dataset\nwith open(r'/kaggle/input/iwildcam-2020-fgvc7/iwildcam2020_train_annotations.json') as train:\n    train_data = json.load(train)\n\n# Testing Dataset\nwith open(r'/kaggle/input/iwildcam-2020-fgvc7/iwildcam2020_test_information.json') as test:\n    test_data = json.load(test)\n    \nprint(\"Columns in training Json: \", train_data.keys())\nprint(\"Columns in testing Json:  \", test_data.keys())","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['categories']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_jpeg = glob.glob('../input/iwildcam-2020-fgvc7/train/*')\ntest_jpeg = glob.glob('../input/iwildcam-2020-fgvc7/test/*')\n\nprint(\"number of train jpeg data:\", len(train_jpeg))\nprint(\"number of test jpeg data:\", len(test_jpeg))\n\ntrain_jpeg[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_show = 15\ncolumns = 5","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CLAHE & Simple WB","metadata":{}},{"cell_type":"code","source":"clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(16, 16))\nwb = cv2.cv2.xphoto.createSimpleWB()\nwb.setP(0.4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img_balance(temp_img):        \n    img_wb = wb.balanceWhite(temp_img)\n\n    img_lab = cv2.cvtColor(img_wb, cv2.COLOR_BGR2Lab)\n\n    l, a, b = cv2.split(img_lab)\n    res_l = clahe.apply(l)\n    res = cv2.merge((res_l, a, b))\n\n    res = cv2.cvtColor(res, cv2.COLOR_Lab2BGR)\n    \n    return res  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Images After Balancing","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(25,12))\nfor idx, train_img in enumerate(train_jpeg):\n    if idx >= num_show:\n        break\n    \n    temp_img = cv2.imread(train_img, cv2.IMREAD_COLOR)        \n\n    res = img_balance(temp_img)\n\n    plt.subplot(10 / columns + 1, columns, idx + 1)\n    plt.imshow(res)\n    if idx % 5 == 4:\n        plt.show()\n        plt.figure(figsize=(25,12))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing Images After Balancing","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(25,12))\nfor idx, train_img in enumerate(test_jpeg):\n    if idx >= num_show:\n        break\n    \n    temp_img = cv2.imread(train_img, cv2.IMREAD_COLOR)        \n\n    res = img_balance(temp_img)\n\n    plt.subplot(10 / columns + 1, columns, idx + 1)\n    plt.imshow(res)\n    if idx % 5 == 4:\n        plt.show()\n        plt.figure(figsize=(25,12))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compare Original and Pre-processed with CLAHE & SimpleWB","metadata":{}},{"cell_type":"code","source":"fig=plt.figure(figsize=(32, 128))\nfor idx, train_img in enumerate(train_jpeg):\n    if idx >= num_show:\n        break\n    \n    temp_img = cv2.imread(train_img, cv2.IMREAD_COLOR)        \n    res = img_balance(temp_img)\n    \n    fig.add_subplot(15, 2, 2 * idx + 1)\n    plt.imshow(temp_img)\n    \n    fig.add_subplot(15, 2, 2 * idx + 2)\n    plt.imshow(res)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of train images: \", len(train_jpeg))\nprint(\"Number of test images: \", len(test_jpeg))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.DataFrame({'id': [item['id'] for item in train_data['annotations']],\n                                'category_id': [item['category_id'] for item in train_data['annotations']],\n                                'image_id': [item['image_id'] for item in train_data['annotations']],\n                                'file_name': [item['file_name'] for item in train_data['images']]})\n\n\ndf_test = pd.DataFrame.from_records(test_data['images'])\n\ndf_train.to_csv('train_data.csv')\ndf_test.to_csv('test_data.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_image = pd.DataFrame.from_records(train_data['images'])\nprint(df_image.head())\n\nindices = []\nfor _id in df_image[df_image['location'] == 537]['id'].values:\n    indices.append( df_train[ df_train['image_id'] == _id ].index )\n\nfor the_index in indices:\n    df_train = df_train.drop(df_train.index[the_index])\n    \ndf_train.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Remove the Unvalid or Corrupt Images from the training data","metadata":{}},{"cell_type":"code","source":"df_train['category_id'] = df_train['category_id'].astype(str)\ndf_train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_CLASSES = len(df_train['category_id'].unique())\nprint(NUM_CLASSES)\n\nnb_classes = len(train_data['categories'])\nprint(nb_classes)\n\nbatch_size = 64\nIMG_SIZE = 64\n\nNUM_EPOCHS = 10\n\nID_COLNAME = 'file_name'\nANSWER_COLNAME = 'category_id'\nTRAIN_IMGS_DIR = r'../input/iwildcam-2020-fgvc7/train/'\nTEST_IMGS_DIR = r'../input/iwildcam-2020-fgvc7/test/'\n\nCHANNELS = 3\n\nIMAGE_RESIZE = 224\nRESNET50_POOLING_AVERAGE = 'avg'\n\nSTEPS_PER_EPOCH_TRAINING = 10\nSTEPS_PER_EPOCH_VALIDATION = 10\n\nBATCH_SIZE_TRAINING = 100\nBATCH_SIZE_VALIDATION = 100\n\n# Using 1 to easily manage mapping between test_generator & prediction for submission preparation\nBATCH_SIZE_TESTING = 1\n\nresnet_weights_path = '../input/resnet50/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_file_cat = df_train[['image_id', 'file_name', 'category_id']]\ndf_train_file_cat['category_id']=df_train_file_cat['category_id'].astype(str)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from PIL import Image\nfrom numpy import asarray\nfrom matplotlib import image\ndef image_pre(img_arr):\n    image = Image.fromarray(img_arr)\n    image0 = img_balance(cv2.imread(image, cv2.IMREAD_COLOR))\n    return image0\nimage0.shape","metadata":{}},{"cell_type":"code","source":"train_datagen=ImageDataGenerator(rescale=1./255, \n                                 validation_split=0.25,\n                                 horizontal_flip = True,    \n                                 zoom_range = 0.3,\n                                 width_shift_range = 0.3,\n                                 height_shift_range=0.3)\n                                 #preprocessing_function = image_pre)\n\ntrain_generator=train_datagen.flow_from_dataframe(    \n    dataframe=df_train[:50000],    \n    directory=\"../input/iwildcam-2020-fgvc7/train\",\n    x_col=ID_COLNAME,\n    y_col=ANSWER_COLNAME,\n    batch_size=batch_size,\n    shuffle=True,\n    classes = [ str(i) for i in range(nb_classes-1)],\n    class_mode=\"categorical\",    \n    target_size=(IMAGE_RESIZE,IMAGE_RESIZE))\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n                                  #preprocessing_function= image_pre)\n\nvalid_generator=test_datagen.flow_from_dataframe(    \n    dataframe=df_train[50000:],    \n    directory=\"../input/iwildcam-2020-fgvc7/train\",\n    x_col=ID_COLNAME,\n    y_col=ANSWER_COLNAME,\n    batch_size=batch_size,\n    shuffle=True,\n    classes = [ str(i) for i in range(nb_classes-1)],\n    class_mode=\"categorical\",  \n    target_size=(IMAGE_RESIZE,IMAGE_RESIZE))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# VGG16 Pretrained Model","metadata":{}},{"cell_type":"code","source":"from keras.applications.vgg16 import VGG16 \nfrom keras_applications.resnet import ResNet50 \nfrom keras.models import Sequential \nfrom keras.layers import Flatten, Dense, Dropout \nfrom keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping\n\nvgg16_model = VGG16(weights=\"imagenet\", include_top=False, input_shape=(224,224,3))\n\nmodel = Sequential()\nmodel.add(vgg16_model)\nmodel.add(Flatten())\nmodel.add(Dense(266, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(266, activation='softmax'))\nmodel.summary()\n\n# Compile model with Adam Optimizer\nmodel.compile(optimizer=Adam(),loss='binary_crossentropy', metrics=['acc'])\n\n\nearly = EarlyStopping(monitor='val_loss', min_delta=0, patience=3, verbose=1, mode='auto')\n\n\nfit_history = model.fit_generator( \n    train_generator, \n    steps_per_epoch=STEPS_PER_EPOCH_TRAINING, \n    epochs = NUM_EPOCHS,\n    validation_data=valid_generator,\n    validation_steps=STEPS_PER_EPOCH_VALIDATION\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(fit_history.history) \nhistory_df[['loss', 'val_loss']].plot() \nhistory_df[['acc', 'val_acc']].plot()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}