{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Original Notebook: [https://www.kaggle.com/aleksandradeis/iwildcam-eda](http://)","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport pandas as pd\nimport numpy as np\nimport datetime as datetime\nimport matplotlib.pyplot as plt\nfrom PIL import Image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-05T06:27:28.192458Z","iopub.execute_input":"2021-09-05T06:27:28.192818Z","iopub.status.idle":"2021-09-05T06:27:28.197121Z","shell.execute_reply.started":"2021-09-05T06:27:28.192777Z","shell.execute_reply":"2021-09-05T06:27:28.196155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setup the directories\nDATA_DIR = '../input/iwildcam2021-fgvc8/'\nTRAIN_DIR = DATA_DIR + 'train/'\nTEST_DIR = DATA_DIR + 'test/'\nMETADATA_DIR = DATA_DIR + 'metadata/'\n\n# load the megadetector results\nmegadetector_results = json.load(open(METADATA_DIR + 'iwildcam2021_megadetector_results.json'))\n#megadetector_results['images'][:2]\n\n# load train images annotations\ntrain_info = json.load(open(METADATA_DIR + 'iwildcam2021_train_annotations.json'))\n# split json into several pandas dataframes\ntrain_annotations = pd.DataFrame(train_info['annotations'])\ntrain_images = pd.DataFrame(train_info['images'])\ntrain_categories = pd.DataFrame(train_info['categories'])\n\n# load test images info\ntest_info = json.load(open(METADATA_DIR + 'iwildcam2021_test_information.json'))\n# split json into several pandas dataframes\ntest_images = pd.DataFrame(test_info['images'])\n#test_categories = pd.DataFrame(test_info['categories'])","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:27:30.428955Z","iopub.execute_input":"2021-09-05T06:27:30.429500Z","iopub.status.idle":"2021-09-05T06:27:35.864275Z","shell.execute_reply.started":"2021-09-05T06:27:30.429454Z","shell.execute_reply":"2021-09-05T06:27:35.863353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_info.keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:27:37.438003Z","iopub.execute_input":"2021-09-05T06:27:37.438370Z","iopub.status.idle":"2021-09-05T06:27:37.446984Z","shell.execute_reply.started":"2021-09-05T06:27:37.438337Z","shell.execute_reply":"2021-09-05T06:27:37.445771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images.keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:36.962181Z","iopub.execute_input":"2021-09-05T06:28:36.962539Z","iopub.status.idle":"2021-09-05T06:28:36.968646Z","shell.execute_reply.started":"2021-09-05T06:28:36.962510Z","shell.execute_reply":"2021-09-05T06:28:36.967706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_annotations.keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:28:39.361254Z","iopub.execute_input":"2021-09-05T06:28:39.361848Z","iopub.status.idle":"2021-09-05T06:28:39.368259Z","shell.execute_reply.started":"2021-09-05T06:28:39.361802Z","shell.execute_reply":"2021-09-05T06:28:39.367404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_categories.keys()\ntrain_categories","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:30:08.527373Z","iopub.execute_input":"2021-09-05T06:30:08.527711Z","iopub.status.idle":"2021-09-05T06:30:08.552777Z","shell.execute_reply.started":"2021-09-05T06:30:08.527683Z","shell.execute_reply":"2021-09-05T06:30:08.551664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_info.keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:44.97001Z","iopub.execute_input":"2021-09-04T06:51:44.970426Z","iopub.status.idle":"2021-09-04T06:51:44.984258Z","shell.execute_reply.started":"2021-09-04T06:51:44.970382Z","shell.execute_reply":"2021-09-04T06:51:44.983184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:44.985932Z","iopub.execute_input":"2021-09-04T06:51:44.986282Z","iopub.status.idle":"2021-09-04T06:51:45.006009Z","shell.execute_reply.started":"2021-09-04T06:51:44.986253Z","shell.execute_reply":"2021-09-04T06:51:45.004831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_categories.head() #there is nothing like in the test json file, it is created for future ","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:45.00764Z","iopub.execute_input":"2021-09-04T06:51:45.008137Z","iopub.status.idle":"2021-09-04T06:51:45.017572Z","shell.execute_reply.started":"2021-09-04T06:51:45.008014Z","shell.execute_reply":"2021-09-04T06:51:45.016653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of images in the train set is {}'.format(train_annotations.image_id.nunique()))\nprint('Number of images in the test set is {}'.format(test_images.file_name.nunique()))","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:45.018779Z","iopub.execute_input":"2021-09-04T06:51:45.019672Z","iopub.status.idle":"2021-09-04T06:51:45.2169Z","shell.execute_reply.started":"2021-09-04T06:51:45.019628Z","shell.execute_reply":"2021-09-04T06:51:45.215853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.pie([train_annotations.image_id.nunique(), test_images.file_name.nunique()], labels=['Train', 'Test'], autopct='%1.1f%%', \n           startangle=90, colors=['#fa4252', '#91bd3a'])\nplt.axis('equal')\nplt.title('Number of images in train and test sets', fontsize=14, color='violet')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:45.218274Z","iopub.execute_input":"2021-09-04T06:51:45.218868Z","iopub.status.idle":"2021-09-04T06:51:45.492836Z","shell.execute_reply.started":"2021-09-04T06:51:45.218823Z","shell.execute_reply":"2021-09-04T06:51:45.492089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Location Data Exploration","metadata":{}},{"cell_type":"code","source":"print('The number of unique locations is {}'.format(train_images.location.nunique()))\nprint('The average number of images per location is {}'.format(train_images.groupby(by=['location']).id.count().mean()))\nprint('The minimum number of images per location is {}'.format(train_images.groupby(by=['location']).id.count().min()))\nprint('The maximum number of images per location is {}'.format(train_images.groupby(by=['location']).id.count().max()))","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:45.493875Z","iopub.execute_input":"2021-09-04T06:51:45.494286Z","iopub.status.idle":"2021-09-04T06:51:45.586201Z","shell.execute_reply.started":"2021-09-04T06:51:45.494242Z","shell.execute_reply":"2021-09-04T06:51:45.58539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\nplt.hist(train_images.groupby(by=['location']).id.count(), bins=40, color='#91bd3a')\nplt.title('The distribution of the number of the images per location', fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:45.587402Z","iopub.execute_input":"2021-09-04T06:51:45.587853Z","iopub.status.idle":"2021-09-04T06:51:45.840314Z","shell.execute_reply.started":"2021-09-04T06:51:45.587813Z","shell.execute_reply":"2021-09-04T06:51:45.839573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Timeline for Captured Images","metadata":{}},{"cell_type":"code","source":"# convert datetimes to just dates\ndef to_date(datetime_str):\n    \"\"\"Convert datetime string to date.\"\"\"\n    # datetime string example: 2013-08-08 11:45:00.000\n    dt = datetime_str.split(' ')[0]\n    return dt\n    \ntrain_images['date'] = train_images.apply(lambda row: to_date(row.datetime), axis=1)\n# group by date\nimg_per_date = train_images.groupby(by=['date']).id.count()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:45.841478Z","iopub.execute_input":"2021-09-04T06:51:45.841928Z","iopub.status.idle":"2021-09-04T06:51:49.460286Z","shell.execute_reply.started":"2021-09-04T06:51:45.841884Z","shell.execute_reply":"2021-09-04T06:51:49.45923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The average number of images per day is {}'.format(img_per_date.mean()))\nprint('The maximum number of images per day is {}'.format(img_per_date.max()))\nprint('The minimum number of images per day is {}'.format(img_per_date.min()))","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:51:49.461609Z","iopub.execute_input":"2021-09-04T06:51:49.461941Z","iopub.status.idle":"2021-09-04T06:51:49.467795Z","shell.execute_reply.started":"2021-09-04T06:51:49.46191Z","shell.execute_reply":"2021-09-04T06:51:49.46703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analyze the number of sequences","metadata":{}},{"cell_type":"code","source":"train_images.keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:55:23.218939Z","iopub.execute_input":"2021-09-04T06:55:23.219313Z","iopub.status.idle":"2021-09-04T06:55:23.225641Z","shell.execute_reply.started":"2021-09-04T06:55:23.219284Z","shell.execute_reply":"2021-09-04T06:55:23.224487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# group by sequence id\nframes_per_sequence = train_images.groupby(by=['seq_id']).seq_frame_num.max()\n\nprint('The average number of frames is {}'.format(frames_per_sequence.mean()))\nprint('The minimum number of frames is {}'.format(frames_per_sequence.min()))\nprint('The maximum number of frames is {}'.format(frames_per_sequence.max()))","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:53:25.599554Z","iopub.execute_input":"2021-09-04T06:53:25.600221Z","iopub.status.idle":"2021-09-04T06:53:25.739077Z","shell.execute_reply.started":"2021-09-04T06:53:25.600186Z","shell.execute_reply":"2021-09-04T06:53:25.737974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(frames_per_sequence.values, bins=40, color='#91bd3a')\nplt.title('The distribution of the number of frames')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T06:55:46.652994Z","iopub.execute_input":"2021-09-04T06:55:46.653486Z","iopub.status.idle":"2021-09-04T06:55:46.859434Z","shell.execute_reply.started":"2021-09-04T06:55:46.653457Z","shell.execute_reply":"2021-09-04T06:55:46.858508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Dimensional Exploration","metadata":{}},{"cell_type":"code","source":"print('The minimum width of the images is {}'.format(train_images.width.min()))\nprint('The maximum width of the images is {}'.format(train_images.width.max()))\nprint('The minimum height of the images is {}'.format(train_images.height.min()))\nprint('The maximum height of the images is {}'.format(train_images.height.max()))","metadata":{"execution":{"iopub.status.busy":"2021-09-04T07:18:53.117039Z","iopub.execute_input":"2021-09-04T07:18:53.117386Z","iopub.status.idle":"2021-09-04T07:18:53.126444Z","shell.execute_reply.started":"2021-09-04T07:18:53.117358Z","shell.execute_reply":"2021-09-04T07:18:53.125358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot histograms to show the distribution of width and height values\nfig, axs = plt.subplots(1, 2, figsize=(15,7))\naxs[0].hist(train_images.width.values, bins=20, color = '#91bd3a')\naxs[0].set_title('Width distribution')\naxs[0].set_xlim(1000, 3000)\n\naxs[1].hist(train_images.width.values, bins=20, color = '#91bd3a')\naxs[1].set_title('Height distribution')\naxs[1].set_xlim(1000, 3000)\n\nplt.suptitle('Image Dimensions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T07:22:17.551521Z","iopub.execute_input":"2021-09-04T07:22:17.552071Z","iopub.status.idle":"2021-09-04T07:22:17.913546Z","shell.execute_reply.started":"2021-09-04T07:22:17.552039Z","shell.execute_reply":"2021-09-04T07:22:17.912681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Set Imnages Exploration","metadata":{}},{"cell_type":"code","source":"def get_first_category(img_id):\n    \"\"\"Find first the image category by id.\"\"\"\n    # get category id\n    category_id = train_annotations[train_annotations.image_id == img_id].category_id.values[0]\n    # get category name\n    category_name = train_categories[train_categories.id == category_id].name.values[0]\n    return category_id, category_name\n\ndef visualize_image_grid(rows, cols):\n    \"\"\"Visualize random grid of images with the first category.\"\"\"\n    filenames = train_images.file_name.unique()\n    \n    np.random.seed(42)\n    img_idx = np.random.randint(len(filenames), size=rows * cols)\n    \n    fig, axs = plt.subplots(rows, cols, figsize=(15,7))\n    \n    for r in range(rows):\n        for c in range(cols):\n            # get the image and image id\n            filename = filenames[img_idx[rows*r + c]]\n            img_id = filename.split('.')[0]\n            # get the category\n            category_id, category = get_first_category(img_id)\n            \n            img = Image.open(TRAIN_DIR + filename)\n            \n            axs[r,c].imshow(img)\n            axs[r,c].axis('off')\n            axs[r,c].set_title('{}:{}'.format(category_id, category))\n            \n    plt.suptitle('Train images', fontsize=16)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T07:24:05.798379Z","iopub.execute_input":"2021-09-04T07:24:05.798909Z","iopub.status.idle":"2021-09-04T07:24:05.810467Z","shell.execute_reply.started":"2021-09-04T07:24:05.798857Z","shell.execute_reply":"2021-09-04T07:24:05.809131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_image_grid(3, 3)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T07:24:07.687325Z","iopub.execute_input":"2021-09-04T07:24:07.687676Z","iopub.status.idle":"2021-09-04T07:24:11.780894Z","shell.execute_reply.started":"2021-09-04T07:24:07.687646Z","shell.execute_reply":"2021-09-04T07:24:11.779595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Specific Image Category Visualization","metadata":{}},{"cell_type":"code","source":"def visualize_cetagory(category_id, rows=3, cols=3, seed=42):\n    \"\"\"Function to visualize images of a specific category.\"\"\"\n    # filter by the category_id\n    copy = train_annotations[train_annotations.category_id == category_id]\n    # get the category name\n    category_name = train_categories[train_categories.id == category_id].name.values[0]\n    \n    # get random indices\n    np.random.seed(seed)\n    img_idx = np.random.randint(len(copy), size=rows * cols)\n    \n    # plot images\n    fig, axs = plt.subplots(rows, cols, figsize=(15,7))\n    \n    for r in range(rows):\n        for c in range(cols):\n            # get the image and image id\n            filename = copy.iloc[img_idx[rows*r + c]].image_id + '.jpg'\n            img_id = filename.split('.')[0]\n            \n            img = Image.open(TRAIN_DIR + filename)\n            \n            axs[r,c].imshow(img)\n            axs[r,c].axis('off')\n            axs[r,c].set_title('{}:{}'.format(category_id, category_name))\n            \n    plt.suptitle('Train images for {}:{}'.format(category_id, category_name), fontsize=16)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T07:34:52.743Z","iopub.execute_input":"2021-09-04T07:34:52.74336Z","iopub.status.idle":"2021-09-04T07:34:52.752859Z","shell.execute_reply.started":"2021-09-04T07:34:52.743329Z","shell.execute_reply":"2021-09-04T07:34:52.751724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_annotations['category_id']","metadata":{"execution":{"iopub.status.busy":"2021-09-04T07:37:17.705385Z","iopub.execute_input":"2021-09-04T07:37:17.705744Z","iopub.status.idle":"2021-09-04T07:37:17.713674Z","shell.execute_reply.started":"2021-09-04T07:37:17.705716Z","shell.execute_reply":"2021-09-04T07:37:17.712637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_cetagory(112) #any number for cateroy id","metadata":{"execution":{"iopub.status.busy":"2021-09-04T07:38:33.116627Z","iopub.execute_input":"2021-09-04T07:38:33.117135Z","iopub.status.idle":"2021-09-04T07:38:36.088881Z","shell.execute_reply.started":"2021-09-04T07:38:33.117093Z","shell.execute_reply":"2021-09-04T07:38:36.088095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize images for top categories","metadata":{}},{"cell_type":"code","source":"# load train images annotations\ntrain_info = json.load(open(METADATA_DIR + 'iwildcam2021_train_annotations.json'))\n# split json into several pandas dataframes\ntrain_annotations = pd.DataFrame(train_info['annotations'])\ntrain_images = pd.DataFrame(train_info['images'])\ntrain_categories = pd.DataFrame(train_info['categories'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import collections\nimport seaborn as sns\n# Preperation for visualization\ndf_categories = pd.DataFrame(train_info[\"categories\"])\nlabels_id = [item[\"id\"] for item in train_info[\"categories\"]]\ncnt = collections.Counter([item[\"category_id\"] for item in train_info[\"annotations\"]])\ndf_categories_count = pd.DataFrame.from_dict(cnt, orient='index').reset_index()\ndf_categories_count = df_categories_count.rename(columns={'index':'id', 0:'count'})\n\ndf_categories_count = df_categories_count.merge(df_categories, on='id').sort_values(by=['count'], ascending=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:54:43.803472Z","iopub.execute_input":"2021-09-05T06:54:43.803962Z","iopub.status.idle":"2021-09-05T06:54:44.771672Z","shell.execute_reply.started":"2021-09-05T06:54:43.803914Z","shell.execute_reply":"2021-09-05T06:54:44.770722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(30, 4))\nax = sns.barplot(x=\"id\", y=\"count\",data=df_categories_count, order=labels_id)\nax.set(ylabel='count')\nax.set(ylim=(0,80000))\nplt.title('distribution of count per id in train')","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:54:46.509868Z","iopub.execute_input":"2021-09-05T06:54:46.510215Z","iopub.status.idle":"2021-09-05T06:54:49.433967Z","shell.execute_reply.started":"2021-09-05T06:54:46.510182Z","shell.execute_reply":"2021-09-05T06:54:49.433276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\nfig = px.bar(df_categories_count, x=\"id\", y=\"count\", \n             title='distribution of count per id in train',\n             width=1400, height=400, color='id')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:58:07.032259Z","iopub.execute_input":"2021-09-05T06:58:07.032651Z","iopub.status.idle":"2021-09-05T06:58:07.093432Z","shell.execute_reply.started":"2021-09-05T06:58:07.032615Z","shell.execute_reply":"2021-09-05T06:58:07.092476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The annotation data seems to be biased to some extent. To see the breakdown, let's look at the top 10 categories. Empty is the most, but annotations stating that animals are in the picture also seem to vary among the top 10.","metadata":{}},{"cell_type":"code","source":"df_categories_count.iloc[:10]","metadata":{"execution":{"iopub.status.busy":"2021-09-05T06:59:16.243742Z","iopub.execute_input":"2021-09-05T06:59:16.244088Z","iopub.status.idle":"2021-09-05T06:59:16.255725Z","shell.execute_reply.started":"2021-09-05T06:59:16.244055Z","shell.execute_reply":"2021-09-05T06:59:16.254538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"On the other hand, fewer categories have only about one sample. We need to be careful when splitting the dataset to train and validation data when training the model.","metadata":{}},{"cell_type":"code","source":"df_categories_count.iloc[-10:]\n","metadata":{"execution":{"iopub.status.busy":"2021-09-05T07:02:26.271668Z","iopub.execute_input":"2021-09-05T07:02:26.272015Z","iopub.status.idle":"2021-09-05T07:02:26.283413Z","shell.execute_reply.started":"2021-09-05T07:02:26.271984Z","shell.execute_reply":"2021-09-05T07:02:26.282457Z"},"trusted":true},"execution_count":null,"outputs":[]}]}