{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport cv2\nimport os\nimport json\n\nfrom PIL import Image, ImageFile, ImageDraw\nImageFile.LOAD_TRUNCATED_IMAGES = True\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, Flatten, Dropout, MaxPooling2D, MaxPool1D\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Set your own project id here\nPROJECT_ID = 'your-google-cloud-project'\nfrom google.cloud import storage\nstorage_client = storage.Client(project=PROJECT_ID)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Use this cell to modify files in output directory\n#os.remove(\"/kaggle/working/train_cropped/905a3c8c-21bc-11ea-a13a-137349068a90.jpg\")\nos.mkdir(\"/kaggle/working/cropped\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with open('../input/iwildcam-2020-fgvc7/iwildcam2020_megadetector_results.json', encoding='utf-8') as json_file:\n    megadetector_results =json.load(json_file)\n    \nmegadetector_results.keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"megadetector_results_df = pd.DataFrame(megadetector_results[\"images\"])\nmegadetector_results_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def draw_bboxs(detections_list, im):\n    \"\"\"\n    detections_list: list of set includes bbox.\n    im: image read by Pillow.\n    \"\"\"\n    \n    for detection in detections_list:\n        x1, y1,w_box, h_box = detection[\"bbox\"]\n        ymin,xmin,ymax, xmax=y1, x1, y1 + h_box, x1 + w_box\n        draw = ImageDraw.Draw(im)\n        \n        imageWidth=im.size[0]\n        imageHeight= im.size[1]\n        (left, right, top, bottom) = (xmin * imageWidth, xmax * imageWidth,\n                                      ymin * imageHeight, ymax * imageHeight)\n        \n        draw.line([(left, top), (left, bottom), (right, bottom),\n               (right, top), (left, top)], width=4, fill='Red')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def cut_bboxs(detections_list, im):\n    \"\"\"\n    detections_list: list of set includes bbox.\n    im: image read by Pillow.\n    \"\"\"\n    \n    #initialize coords as entire image  \n    imageWidth=im.size[0]\n    imageHeight= im.size[1]\n    \n    top_left_coord = [imageWidth, imageHeight]\n    bottom_right_coord = [0,0]\n    #print(top_left_coord[0],top_left_coord[1], bottom_right_coord[0], bottom_right_coord[1])\n\n    if len(detections_list) == 0:\n        return im\n    \n    for detection in detections_list:\n        x1, y1,w_box, h_box = detection[\"bbox\"]\n        ymin,xmin,ymax, xmax=y1*imageHeight, x1*imageWidth, (y1 + h_box)*imageHeight, (x1 + w_box)*imageWidth\n        #print(ymin,xmin,ymax, xmax)\n        if ymax > bottom_right_coord[1]:\n            bottom_right_coord[1] = ymax\n        if xmax > bottom_right_coord[0]:\n            bottom_right_coord[0] = xmax\n        if ymin < top_left_coord[1]:\n            top_left_coord[1] = ymin\n        if xmin < top_left_coord[0]:\n            top_left_coord[0] = xmin\n    \n    draw = ImageDraw.Draw(im)\n    \n    \n    (left, right, top, bottom) = (top_left_coord[0] , bottom_right_coord[0] ,\n                                  top_left_coord[1] , bottom_right_coord[1] )\n    #print(left, top, right, bottom)\n    new_img = im.crop((left, top, right, bottom))\n    return new_img","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# setup the directories\nDATA_DIR = '../input/iwildcam-2020-fgvc7/'\nTRAIN_DIR = DATA_DIR + 'train/'\nTEST_DIR = DATA_DIR + 'test/'\n\n# load the megadetector results\nmegadetector_results = json.load(open(DATA_DIR + 'iwildcam2020_megadetector_results.json'))\nprint(megadetector_results['images'][:2])\n\n# load train images annotations\ntrain_info = json.load(open(DATA_DIR + 'iwildcam2020_train_annotations.json'))\n# split json into several pandas dataframes\ntrain_annotations = pd.DataFrame(train_info['annotations'])\ntrain_images = pd.DataFrame(train_info['images'])\ntrain_categories = pd.DataFrame(train_info['categories'])\n\n# load test images info\ntest_info = json.load(open(DATA_DIR + 'iwildcam2020_test_information.json'))\n# split json into several pandas dataframes\ntest_images = pd.DataFrame(test_info['images'])\ntest_categories = pd.DataFrame(test_info['categories'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(train_images))\ntrain_length = len(megadetector_results_df)\n#print(train_images.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"missing = []\nfor i in range(20):\n    try:\n        data_num = i\n        try:\n            im = Image.open(\"../input/iwildcam-2020-fgvc7/train/\" + megadetector_results_df.loc[data_num]['id'] + \".jpg\")\n        except:\n            im = Image.open(\"../input/iwildcam-2020-fgvc7/test/\" + megadetector_results_df.loc[data_num]['id'] + \".jpg\")\n        #im = im.resize((500,325))\n        new_img = cut_bboxs(megadetector_results_df.loc[data_num]['detections'], im)\n        draw_bboxs(megadetector_results_df.loc[data_num]['detections'], im)\n        new_img = new_img.resize((500,325))\n        #new_img.save(\"test2.jpeg\")\n        #new_img.save(\"./cropped/\" + megadetector_results_df.loc[data_num]['id'] + \".jpg\")\n        plt.imshow(im)\n        plt.show()\n        plt.imshow(new_img)\n        plt.show()\n        #os.system('cls')\n        #print(train_length - i)\n        #print(len(missing))\n    except:\n        missing.append(i)\n        #print(\"Error at:\", i)\n        os.system('cls')\n        print(train_length - i)\n        print(len(missing))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len([name for name in os.listdir('./train_cropped/') if os.path.isfile(name)]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.set_option('display.max_rows', 500)\ntrain_categories","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.merge(train_annotations, train_images, how='outer', left_on='image_id', right_on='id')\ndf_train = df_train.drop(['id_y'], axis=1)\n\ndf_train['Date'] = pd.to_datetime(df_train['datetime']).dt.date\ndf_train = df_train.astype({'Date': str})\ndf_train['Time'] = pd.to_datetime(df_train['datetime']).dt.time\ndf_train = df_train.astype({'Time': str})\n\ndf_train['Year'] = df_train['Date'].str.slice(0, 4, 1)\ndf_train = df_train.astype({'Year': int})\ndf_train['Month'] = df_train['Date'].str.slice(5, 7, 1)\ndf_train = df_train.astype({'Month': int})\ndf_train['Day'] = df_train['Date'].str.slice(8, 10, 1)\ndf_train = df_train.astype({'Day': int})\n\ndf_train['Hour'] = df_train['Time'].str.slice(0, 2, 1)\ndf_train = df_train.astype({'Hour': int})\ndf_train['Min'] = df_train['Time'].str.slice(3, 5, 1)\ndf_train = df_train.astype({'Min': int})\ndf_train['Sec'] = df_train['Time'].str.slice(6, 8, 1)\ndf_train = df_train.astype({'Sec': int})\n\ndf_train = df_train.drop(['Date', 'Time'], axis=1)\n\ndf_train.columns = ['animal_cnt', 'image_id', 'id', 'category_id', 'seq_num_frames', 'location', 'datetime', 'frame_num', 'seq_id', 'width', 'height', 'file_name', 'year',\n                    'month', 'day', 'hour', 'min', 'sec']\n\ndf_train = df_train[['id', 'seq_id', 'image_id', 'file_name', 'width', 'height', 'seq_num_frames', 'frame_num', 'datetime', 'location', 'animal_cnt', 'year',\n                     'month', 'day', 'hour', 'min', 'sec', 'category_id']]\n\ndf_train['category_id'] = df_train['category_id'].apply(str)\n\ndf_train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 16\nepochs = 1\nIMG_HEIGHT = 325\nIMG_WIDTH = 500","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corrupted_files = ['87022118-21bc-11ea-a13a-137349068a90.jpg',\n                   '8f17b296-21bc-11ea-a13a-137349068a90.jpg',\n                   '883572ba-21bc-11ea-a13a-137349068a90.jpg',\n                   '896c1198-21bc-11ea-a13a-137349068a90.jpg',\n                   '8792549a-21bc-11ea-a13a-137349068a90.jpg',\n                   '99136aa6-21bc-11ea-a13a-137349068a90.jpg']\n\nfor corrupted_file in corrupted_files:\n    df_train = df_train[df_train['file_name'] != corrupted_file]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"total_train_image_generator = ImageDataGenerator(rescale=1./255, rotation_range=20, zoom_range=0.15, width_shift_range=0.2, \n                                                 height_shift_range=0.2, shear_range=0.15, horizontal_flip=True, \n                                                 fill_mode=\"nearest\")\ntotal_train_data_gen = total_train_image_generator.flow_from_dataframe(dataframe=df_train, directory=TRAIN_DIR, x_col='file_name', y_col='category_id', \n                                                            class_mode=\"categorical\", target_size=(IMG_HEIGHT,IMG_WIDTH), batch_size=batch_size)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.utils import class_weight\n\nclass_weights = class_weight.compute_class_weight(\n               'balanced',\n                np.unique(total_train_data_gen.classes), \n                total_train_data_gen.classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"total_train_data_gen.class_indices\nREVERSE_CLASSMAP = dict([(v, k) for k, v in total_train_data_gen.class_indices.items()])\nREVERSE_CLASSMAP","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_training_images, _ = next(total_train_data_gen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# This function will plot images in the form of a grid with 1 row and 5 columns where images are placed in each column.\ndef plotImages(images_arr):\n    fig, axes = plt.subplots(1, 5, figsize=(20,20))\n    axes = axes.flatten()\n    for img, ax in zip( images_arr, axes):\n        ax.imshow(img)\n        ax.axis('off')\n    plt.tight_layout()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plotImages(sample_training_images[:5])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential([\n    Conv2D(16, 3, padding='same', activation='relu', input_shape=(IMG_HEIGHT, IMG_WIDTH ,3)),\n    MaxPooling2D(),\n    Conv2D(16, 3, padding='same', activation='relu'),\n    MaxPooling2D(),\n    Conv2D(16, 3, padding='same', activation='relu'),\n    MaxPooling2D(),\n    Flatten(),\n    Dense(256, activation='relu'),\n    Dense(len(total_train_data_gen.class_indices))\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(optimizer='adam',\n              loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(\n    total_train_data_gen,\n    epochs=epochs,\n    class_weight=class_weights\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"acc = history.history['accuracy']\n\nloss = history.history['loss']\n\nepochs_range = range(epochs)\n\nplt.figure(figsize=(8, 8))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.legend(loc='lower right')\nplt.title('TrainingAccuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.legend(loc='upper right')\nplt.title('TrainingLoss')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test = df_train.sample(15)\n\nprobability_model = tf.keras.Sequential([model, tf.keras.layers.Softmax()])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def predict_image(model, file_name, probability_model):\n    img = Image.open(file_name)\n    img = img.resize((IMG_WIDTH, IMG_HEIGHT))\n    test_img = np.array([np.asarray(img)])\n\n    prediction = probability_model.predict(test_img)\n    pred_class = REVERSE_CLASSMAP[np.argmax(prediction)]\n    \n    return pred_class","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_image(pred_label, true_label, img_filename):  \n    img = Image.open(img_filename)\n    img = img.resize((IMG_WIDTH, IMG_HEIGHT))\n\n    plt.grid(False)\n    plt.xticks([])\n    plt.yticks([])\n\n    plt.imshow(img, cmap=plt.cm.binary)\n\n    if pred_label == true_label:\n        color = 'blue'\n    else:\n        color = 'red'\n    \n    plt.title(\"{} ({})\".format(train_categories.loc[train_categories['id'] == int(true_label), 'name'].to_string(index=False),\n                                true_label))\n    \n    plt.xlabel(\"{} ({})\".format(train_categories.loc[train_categories['id'] == int(pred_label), 'name'].to_string(index=False),\n                                pred_label)\n               ,color=color)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"count = 0\nfor filename in df_test['file_name']:\n    pred_class = predict_image(model, TRAIN_DIR + filename, probability_model)\n    \n    plot_image(pred_class, df_test['category_id'].iloc[count], TRAIN_DIR + filename)\n    \n    count += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv(DATA_DIR + 'sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"count = 0\nfor ids in submission['Id']:\n    if count % 1000 == 0:\n        print('Through ' + str(count) + ' images')\n    \n    filename = TEST_DIR + ids + '.jpg'\n    pred_cat = predict_image(model, filename, probability_model)\n    \n    submission['Category'][count] = int(pred_cat)\n    \n    count += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('E:/submission_004.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}