{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import json\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom PIL import Image, UnidentifiedImageError\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_annot_path = '../input/iwildcam-2020-fgvc7/iwildcam2020_train_annotations.json'\ntrain_img_path = '../input/iwildcam-2020-fgvc7/train'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with open(train_annot_path) as f:\n    train_annot = json.load(f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'Total categories in set: {len(train_annot[\"categories\"])}')\nprint(f'Total training images: {len(train_annot[\"images\"])}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_labels = pd.DataFrame(train_annot['annotations'])\ntrain_labels.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_with_1_label = np.count_nonzero(train_labels.groupby('image_id').category_id.nunique().values == 1)\nprint('1 label for each image: {}'.format(images_with_1_label == len(train_annot['images'])))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Model trained on this may struggle when multiple animals are in frame"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_classes = train_labels.category_id.nunique()\nprint('All classes have training examples: {}'.format(len(train_annot['categories']) == train_classes))\nprint('Total classes: {}\\nTrain classes: {}'.format(len(train_annot['categories']), train_classes))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Interesting..."},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's look at some images\nfig = plt.figure(figsize=(24, 12))\nfor i, img_id in enumerate(train_labels.image_id.sample(12).values):\n    img_path = os.path.join(train_img_path, img_id + '.jpg')\n    img = Image.open(img_path)\n    ax = fig.add_subplot(3, 4, i+1)\n    ax.imshow(img)\n    ax.grid()\n    ax.axis('off')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images_meta = pd.DataFrame(train_annot['images'])\ntrain_images_meta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images_meta.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There can be multiple images per sequence.\n\nSome images have sides = -1 ???\n\nSignificantly high stddev in image height and width"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images_meta[train_images_meta.seq_num_frames == 10]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There's variance in the size of image within a sequence. Let's look at a sequence"},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_sequence = train_images_meta[train_images_meta.seq_num_frames == 10].seq_id.sample(1).values[0]\nseq_img = train_images_meta[train_images_meta.seq_id == sample_sequence].file_name.values\n\n# Let's look at some images\nfig = plt.figure(figsize=(30, 30))\nfor i, img_id in enumerate(seq_img):\n    img_path = os.path.join(train_img_path, img_id)\n    img = Image.open(img_path)\n    ax = fig.add_subplot(4, 3, i+1)\n    ax.imshow(img)\n    ax.grid()\n    ax.axis('off')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_dims = train_images_meta.groupby(['height', 'width']).id.nunique().reset_index().sort_values(by='id', ascending=False)\nimg_dims['frac'] = img_dims.id / img_dims.id.sum()\nimg_dims['cum_frac'] = img_dims.id.cumsum() / img_dims.id.sum()\nimg_dims.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"~98% images are one of (1024p/HD, 1080p/FHD, 1536p). Decent for downscaling"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nimg_dims = np.log10(train_images_meta.groupby(['height', 'width']).id.nunique()).reset_index()\nimg_dims = pd.pivot_table(img_dims, index='height', columns='width', values='id').fillna(-1)\nsns.heatmap(img_dims, square=True, linecolor='#09000f', linewidths=.1)\n_ = plt.gca().set_title('Image dimensions (counts in log scale)')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Images with -1 dims are intriguing\n# Let's look at some\nfig = plt.figure(figsize=(24, 6))\nfor i, img_id in enumerate(train_images_meta[train_images_meta.width == -1].id.values):\n    try:\n        img_path = os.path.join(train_img_path, img_id + '.jpg')\n        img = Image.open(img_path)\n        ax = fig.add_subplot(1, 3, i+1)\n        ax.imshow(img)\n        ax.grid()\n        ax.axis('off')\n    except FileNotFoundError:\n        print(\"Image {} doesn't exist\".format(img_id))\n    except UnidentifiedImageError:\n        print(\"Image {} unidentified\".format(img_id))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"These don't exist in training dataset. Should remove before training"},{"metadata":{"trusted":true},"cell_type":"code","source":"categories = pd.DataFrame(train_annot['categories'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"examples_by_cat = train_labels.category_id.value_counts().reset_index()\nexamples_by_cat.columns = ['category', 'examples']\nexamples_by_cat = examples_by_cat.merge(categories, left_on='category', right_on='id', how='inner')[['id', 'name', 'examples']]\nexamples_by_cat = examples_by_cat.assign(cumulative=examples_by_cat.examples.cumsum()/examples_by_cat.examples.sum(), frac=examples_by_cat.examples/examples_by_cat.examples.sum())\nexamples_by_cat.head(20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"~35% examples belong to the empty class"},{"metadata":{"trusted":true},"cell_type":"code","source":"examples_by_cat[examples_by_cat.id > 0][['examples', 'frac']].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"examples_by_cat.sort_values('frac')[['name', 'examples']].head(20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Some classes have only a single example"},{"metadata":{"trusted":true},"cell_type":"code","source":"l10_examples = examples_by_cat.groupby('examples').id.count().reset_index()\nl10_examples.columns=['examples', 'n_classes']\nprint('{} classes have less than 10 examples'.format(l10_examples[l10_examples.examples<=10].n_classes.sum()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}