{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-21T14:35:03.399925Z","iopub.execute_input":"2022-07-21T14:35:03.400561Z","iopub.status.idle":"2022-07-21T14:35:03.428618Z","shell.execute_reply.started":"2022-07-21T14:35:03.400452Z","shell.execute_reply":"2022-07-21T14:35:03.427175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show Ground Truth Annotations for Sequences in Training Dataset","metadata":{}},{"cell_type":"code","source":"df_count = pd.read_csv(\"../input/iwildcam2022-fgvc9/metadata/metadata/train_sequence_counts.csv\")\nprint(len(df_count.index))\ndf_count.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:03.439109Z","iopub.execute_input":"2022-07-21T14:35:03.439555Z","iopub.status.idle":"2022-07-21T14:35:03.491649Z","shell.execute_reply.started":"2022-07-21T14:35:03.439523Z","shell.execute_reply":"2022-07-21T14:35:03.490817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plot Distribution of Counts in Sequence**","metadata":{}},{"cell_type":"code","source":"count = []\nfor i in range(max(df_count[\"count\"])):\n    count.append(len(df_count[df_count[\"count\"]==i]))\nf = plt.figure(figsize=(15,8))\nf = plt.title(\"distribution of count\")\nf = plt.bar([x for x in range(len(count))],count)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:03.493944Z","iopub.execute_input":"2022-07-21T14:35:03.494245Z","iopub.status.idle":"2022-07-21T14:35:03.762229Z","shell.execute_reply.started":"2022-07-21T14:35:03.494218Z","shell.execute_reply":"2022-07-21T14:35:03.761221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import, Explore and Preprocess Metadata","metadata":{}},{"cell_type":"code","source":"import json, codecs\n\nwith codecs.open(\"../input/iwildcam2022-fgvc9/metadata/metadata/iwildcam2022_train_annotations.json\", 'r',\n                 encoding='utf-8', errors='ignore') as f:\n    train_meta = json.load(f)\n    print(train_meta.keys())\n    \nwith codecs.open(\"../input/iwildcam2022-fgvc9/metadata/metadata/iwildcam2022_test_information.json\", 'r',\n                 encoding='utf-8', errors='ignore') as f:\n    test_meta = json.load(f)\n    print(test_meta.keys())\n    \nwith codecs.open(\"../input/iwildcam2022-fgvc9/metadata/metadata/iwildcam2022_mdv4_detections.json\", 'r',\n                 encoding='utf-8', errors='ignore') as f:\n    detections = json.load(f)\n    print(detections.keys())","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:03.763699Z","iopub.execute_input":"2022-07-21T14:35:03.764134Z","iopub.status.idle":"2022-07-21T14:35:10.660809Z","shell.execute_reply.started":"2022-07-21T14:35:03.764094Z","shell.execute_reply":"2022-07-21T14:35:10.659595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Extract annotated count sequences**","metadata":{}},{"cell_type":"code","source":"df_trainimages = pd.DataFrame(train_meta['images'])\ndf_trainimages = df_trainimages.drop([\"seq_num_frames\",\"datetime\",\"width\",\"width\",\"height\",\"id\",\"sub_location\",\"location\"],axis=1)\nprint(len(df_trainimages.index))\n#remove sequences which are in df_count\ndf_countedimageseq = df_trainimages[df_trainimages['seq_id'].isin(df_count['seq_id'].tolist())]\ndf_trainimages = df_trainimages[~df_trainimages['seq_id'].isin(df_count['seq_id'].tolist())]\n\nprint(len(df_trainimages.index))\ndf_trainimages.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:36:54.671747Z","iopub.execute_input":"2022-07-21T14:36:54.672127Z","iopub.status.idle":"2022-07-21T14:36:55.474929Z","shell.execute_reply.started":"2022-07-21T14:36:54.672096Z","shell.execute_reply":"2022-07-21T14:36:55.473851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_trainannotations = pd.DataFrame(train_meta['annotations'])\ndf_trainannotations.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:11.477775Z","iopub.execute_input":"2022-07-21T14:35:11.478559Z","iopub.status.idle":"2022-07-21T14:35:11.719517Z","shell.execute_reply.started":"2022-07-21T14:35:11.478526Z","shell.execute_reply":"2022-07-21T14:35:11.718733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plot species distribution**","metadata":{}},{"cell_type":"code","source":"annotations = []\nfor i in range(max(df_trainannotations[\"category_id\"])):\n    annotations.append(len(df_trainannotations[df_trainannotations[\"category_id\"]==i]))\nf = plt.figure(figsize=(15,8))\nf = plt.title(\"distribution of category_id\")\nf = plt.bar([x for x in range(len(annotations))],annotations)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:11.720499Z","iopub.execute_input":"2022-07-21T14:35:11.721360Z","iopub.status.idle":"2022-07-21T14:35:13.189969Z","shell.execute_reply.started":"2022-07-21T14:35:11.721329Z","shell.execute_reply":"2022-07-21T14:35:13.188840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Inspect species**","metadata":{}},{"cell_type":"code","source":"df_seq = pd.DataFrame(train_meta['categories'])\ndf_seq","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:13.191226Z","iopub.execute_input":"2022-07-21T14:35:13.191547Z","iopub.status.idle":"2022-07-21T14:35:13.203808Z","shell.execute_reply.started":"2022-07-21T14:35:13.191518Z","shell.execute_reply":"2022-07-21T14:35:13.202637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**merge training metadata with species annotation**","metadata":{}},{"cell_type":"code","source":"df_trainannotations[\"file_name\"]=df_trainannotations[\"image_id\"].astype(str) +\".jpg\"\ndf_trainimages_trainannotations = pd.merge(df_trainimages, df_trainannotations.drop([\"id\",\"image_id\"],axis=1), on=\"file_name\")\ndf_trainannotations = df_trainannotations.drop([\"file_name\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:13.209402Z","iopub.execute_input":"2022-07-21T14:35:13.209740Z","iopub.status.idle":"2022-07-21T14:35:13.616957Z","shell.execute_reply.started":"2022-07-21T14:35:13.209710Z","shell.execute_reply":"2022-07-21T14:35:13.615661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_trainimages_trainannotations","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:13.618296Z","iopub.execute_input":"2022-07-21T14:35:13.618641Z","iopub.status.idle":"2022-07-21T14:35:13.639812Z","shell.execute_reply.started":"2022-07-21T14:35:13.618612Z","shell.execute_reply":"2022-07-21T14:35:13.638914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Subset for Bounding Boxes","metadata":{}},{"cell_type":"markdown","source":"**subsample with max_samples_per_class**","metadata":{}},{"cell_type":"code","source":"max_samples_per_class=1\nimg_subset = df_trainimages_trainannotations[df_trainimages_trainannotations[\"category_id\"]==0].sample(frac=1,random_state=42).head(max_samples_per_class)\ndf_seq[\"id\"]\nfor i in df_seq[\"id\"].tail(len(df_seq)-1).tolist():\n    img_subset = img_subset.append(df_trainimages_trainannotations[df_trainimages_trainannotations[\"category_id\"]==i].sample(frac=1,random_state=42).head(max_samples_per_class))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:13.641214Z","iopub.execute_input":"2022-07-21T14:35:13.641831Z","iopub.status.idle":"2022-07-21T14:35:14.078459Z","shell.execute_reply.started":"2022-07-21T14:35:13.641796Z","shell.execute_reply":"2022-07-21T14:35:14.077609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**species distribution after sampling**","metadata":{}},{"cell_type":"code","source":"annotations = []\nfor i in range(max(img_subset[\"category_id\"])):\n    annotations.append(len(img_subset[img_subset[\"category_id\"]==i]))\nf = plt.figure(figsize=(15,8))\nf = plt.title(\"distribution of category_id\")\nf = plt.bar([x for x in range(len(annotations))],annotations)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:14.080514Z","iopub.execute_input":"2022-07-21T14:35:14.081362Z","iopub.status.idle":"2022-07-21T14:35:15.417440Z","shell.execute_reply.started":"2022-07-21T14:35:14.081290Z","shell.execute_reply":"2022-07-21T14:35:15.416248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset = img_subset.drop([\"seq_frame_num\",\"seq_id\"],axis=1)\nimg_subset","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:15.420457Z","iopub.execute_input":"2022-07-21T14:35:15.421683Z","iopub.status.idle":"2022-07-21T14:35:15.436633Z","shell.execute_reply.started":"2022-07-21T14:35:15.421632Z","shell.execute_reply":"2022-07-21T14:35:15.435514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_detect = pd.DataFrame(detections[\"images\"])\ndf_detect = df_detect.rename(columns={\"file\": \"file_name\"})\ndf_detect[\"file_name\"] = df_detect[\"file_name\"].astype(str).str.split('/').str[-1]\ndf_detect.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:15.438460Z","iopub.execute_input":"2022-07-21T14:35:15.439282Z","iopub.status.idle":"2022-07-21T14:35:16.702145Z","shell.execute_reply.started":"2022-07-21T14:35:15.439128Z","shell.execute_reply":"2022-07-21T14:35:16.700955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset = img_subset.merge(df_detect, on=\"file_name\")\nimg_subset","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:16.703890Z","iopub.execute_input":"2022-07-21T14:35:16.705230Z","iopub.status.idle":"2022-07-21T14:35:16.882056Z","shell.execute_reply.started":"2022-07-21T14:35:16.705075Z","shell.execute_reply":"2022-07-21T14:35:16.880712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Extract Detections with confidence > 0.5**","metadata":{}},{"cell_type":"code","source":"img_subset_g05 = img_subset.loc[(img_subset['max_detection_conf']>0.5)] #  | (img_subset['max_detection_conf'] == 0)\nimg_subset_g05","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:16.884172Z","iopub.execute_input":"2022-07-21T14:35:16.885019Z","iopub.status.idle":"2022-07-21T14:35:16.940673Z","shell.execute_reply.started":"2022-07-21T14:35:16.884970Z","shell.execute_reply":"2022-07-21T14:35:16.939501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = {'path', 'x', 'y', 'w', \"h\",\"category\"}\nannotation_csv = pd.DataFrame(data=d)\nannotation_csv = pd.DataFrame(columns=['path', 'x', 'y', 'w', \"h\",\"category\"])\nannotation_csv","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:16.942577Z","iopub.execute_input":"2022-07-21T14:35:16.943400Z","iopub.status.idle":"2022-07-21T14:35:16.958966Z","shell.execute_reply.started":"2022-07-21T14:35:16.943352Z","shell.execute_reply":"2022-07-21T14:35:16.957750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Extract Detections -> one row for each detection**","metadata":{}},{"cell_type":"code","source":"import matplotlib.image as mpimg\n\n#construct csv df\nfor index, row in img_subset_g05.iterrows():\n    filename = row[\"file_name\"]\n    img_path=\"../input/iwildcam2022-fgvc9/train/train/\" + row[\"file_name\"]\n    img_id=img_path.split('/')[-1][0:-4]\n    img=mpimg.imread(img_path)\n    H,W,_=img.shape\n    for detection in row[\"detections\"]:\n        category = detection[\"category\"]\n        x = int(detection[\"bbox\"][0]*W)\n        y = int(detection[\"bbox\"][1]*H)\n        w = x +int(detection[\"bbox\"][2]*W)\n        h = y + int(detection[\"bbox\"][3]*H)\n        d = {'path': [filename], 'x': [x], 'y': [y], 'w': [w], \"h\":[h],\"category\": [category]}\n        append = pd.DataFrame(data=d)\n        annotation_csv = annotation_csv.append(append)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:16.960747Z","iopub.execute_input":"2022-07-21T14:35:16.961472Z","iopub.status.idle":"2022-07-21T14:35:28.155735Z","shell.execute_reply.started":"2022-07-21T14:35:16.961424Z","shell.execute_reply":"2022-07-21T14:35:28.154629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**add some samples with no detections**","metadata":{}},{"cell_type":"code","source":"img_subset_0 = img_subset.loc[(img_subset['max_detection_conf'] == 0)]\nfor index, row in img_subset_0.iterrows():\n    d = {'path': row[\"file_name\"], 'x': [], 'y': [], 'w': [], \"h\":[],\"category\": []}\n    append = pd.DataFrame(data=d)\n    annotation_csv = annotation_csv.append(append)\nannotation_csv = annotation_csv.sample(frac=1,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.157275Z","iopub.execute_input":"2022-07-21T14:35:28.158040Z","iopub.status.idle":"2022-07-21T14:35:28.219943Z","shell.execute_reply.started":"2022-07-21T14:35:28.157995Z","shell.execute_reply":"2022-07-21T14:35:28.218748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Generated df for downsampled dataset**","metadata":{}},{"cell_type":"code","source":"annotation_csv","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.221577Z","iopub.execute_input":"2022-07-21T14:35:28.222263Z","iopub.status.idle":"2022-07-21T14:35:28.239400Z","shell.execute_reply.started":"2022-07-21T14:35:28.222216Z","shell.execute_reply":"2022-07-21T14:35:28.238278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Export Data","metadata":{}},{"cell_type":"code","source":"#!mkdir images","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.282475Z","iopub.execute_input":"2022-07-21T14:35:28.283591Z","iopub.status.idle":"2022-07-21T14:35:28.287987Z","shell.execute_reply.started":"2022-07-21T14:35:28.283551Z","shell.execute_reply":"2022-07-21T14:35:28.286891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#move images to output\n#for index, row in img_subset.iterrows():\n#    file_location = \"../input/iwildcam2022-fgvc9/train/train/\" + row[\"file_name\"]\n#    file_target = \"./images/\" + row[\"file_name\"]\n#    !cp \"$file_location\" \"$file_target\"","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.289260Z","iopub.execute_input":"2022-07-21T14:35:28.289616Z","iopub.status.idle":"2022-07-21T14:35:28.297733Z","shell.execute_reply.started":"2022-07-21T14:35:28.289586Z","shell.execute_reply":"2022-07-21T14:35:28.296816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#annotation_csv.to_csv(r'./annotation.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.299168Z","iopub.execute_input":"2022-07-21T14:35:28.299737Z","iopub.status.idle":"2022-07-21T14:35:28.307810Z","shell.execute_reply.started":"2022-07-21T14:35:28.299707Z","shell.execute_reply":"2022-07-21T14:35:28.306898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!zip -r ./images.zip ./images","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.308734Z","iopub.execute_input":"2022-07-21T14:35:28.309044Z","iopub.status.idle":"2022-07-21T14:35:28.317507Z","shell.execute_reply.started":"2022-07-21T14:35:28.309005Z","shell.execute_reply":"2022-07-21T14:35:28.316743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Subset for Counts (classification/regression)","metadata":{}},{"cell_type":"code","source":"df_trainannotations[\"file_name\"]=df_trainannotations[\"image_id\"].astype(str) +\".jpg\"\ndf_countedimageseq_trainannotations = pd.merge(df_countedimageseq, df_trainannotations.drop([\"id\",\"image_id\"],axis=1), on=\"file_name\")\ndf_trainannotations = df_trainannotations.drop([\"file_name\"],axis=1)\ndf_countedimageseq_trainannotations","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.319094Z","iopub.execute_input":"2022-07-21T14:35:28.319501Z","iopub.status.idle":"2022-07-21T14:35:28.540377Z","shell.execute_reply.started":"2022-07-21T14:35:28.319461Z","shell.execute_reply":"2022-07-21T14:35:28.539226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**subsample with max_samples_per_class**","metadata":{}},{"cell_type":"code","source":"max_samples_per_class=500\nimg_subset_classification = df_countedimageseq_trainannotations[df_countedimageseq_trainannotations[\"category_id\"]==0].sample(frac=1,random_state=42).head(max_samples_per_class)\ndf_seq[\"id\"]\nfor i in df_seq[\"id\"].tail(len(df_seq)-1).tolist():\n    img_subset_classification = img_subset_classification.append(df_countedimageseq_trainannotations[df_countedimageseq_trainannotations[\"category_id\"]==i].sample(frac=1,random_state=42).head(max_samples_per_class))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.565514Z","iopub.execute_input":"2022-07-21T14:35:28.565948Z","iopub.status.idle":"2022-07-21T14:35:28.977697Z","shell.execute_reply.started":"2022-07-21T14:35:28.565906Z","shell.execute_reply":"2022-07-21T14:35:28.976616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset_classification","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.979212Z","iopub.execute_input":"2022-07-21T14:35:28.980149Z","iopub.status.idle":"2022-07-21T14:35:28.994407Z","shell.execute_reply.started":"2022-07-21T14:35:28.980105Z","shell.execute_reply":"2022-07-21T14:35:28.993348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**species distribution after sampling**","metadata":{}},{"cell_type":"code","source":"annotations = []\nfor i in range(max(img_subset_classification[\"category_id\"])):\n    annotations.append(len(img_subset_classification[img_subset_classification[\"category_id\"]==i]))\nf = plt.figure(figsize=(15,8))\nf = plt.title(\"distribution of category_id\")\nf = plt.bar([x for x in range(len(annotations))],annotations)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:28.995690Z","iopub.execute_input":"2022-07-21T14:35:28.996017Z","iopub.status.idle":"2022-07-21T14:35:30.292376Z","shell.execute_reply.started":"2022-07-21T14:35:28.995989Z","shell.execute_reply":"2022-07-21T14:35:30.291336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Include detections to generate training label, since annotated count applies for the whole sequence but not for a single image**","metadata":{}},{"cell_type":"code","source":"img_subset_classification = img_subset_classification.merge(df_detect, on=\"file_name\")\nimg_subset_classification","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:30.293832Z","iopub.execute_input":"2022-07-21T14:35:30.294164Z","iopub.status.idle":"2022-07-21T14:35:30.459779Z","shell.execute_reply.started":"2022-07-21T14:35:30.294134Z","shell.execute_reply":"2022-07-21T14:35:30.458548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Filter out images with low max_detection_conf to reduce error propagation**","metadata":{}},{"cell_type":"code","source":"img_subset_classification = img_subset_classification.loc[(img_subset_classification['max_detection_conf'] > 0.5) | (img_subset['max_detection_conf'] == 0)]\nimg_subset_classification","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:30.461839Z","iopub.execute_input":"2022-07-21T14:35:30.462632Z","iopub.status.idle":"2022-07-21T14:35:30.513780Z","shell.execute_reply.started":"2022-07-21T14:35:30.462584Z","shell.execute_reply":"2022-07-21T14:35:30.512823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset_classification = img_subset_classification.drop([\"seq_frame_num\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:30.516231Z","iopub.execute_input":"2022-07-21T14:35:30.516535Z","iopub.status.idle":"2022-07-21T14:35:30.523040Z","shell.execute_reply.started":"2022-07-21T14:35:30.516508Z","shell.execute_reply":"2022-07-21T14:35:30.522071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Count Megadetector Detections**","metadata":{}},{"cell_type":"code","source":"conf_sum_list = []\nfor index, row in img_subset_classification.iterrows():\n    conf_sum = 0\n    for detection in row[\"detections\"]:\n        conf_sum = conf_sum + 1\n    conf_sum_list.append(conf_sum)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:30.524526Z","iopub.execute_input":"2022-07-21T14:35:30.525475Z","iopub.status.idle":"2022-07-21T14:35:30.956745Z","shell.execute_reply.started":"2022-07-21T14:35:30.525433Z","shell.execute_reply":"2022-07-21T14:35:30.955753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset_classification['bbox_count'] = conf_sum_list\nimg_subset_classification","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:30.958011Z","iopub.execute_input":"2022-07-21T14:35:30.958598Z","iopub.status.idle":"2022-07-21T14:35:30.996313Z","shell.execute_reply.started":"2022-07-21T14:35:30.958562Z","shell.execute_reply":"2022-07-21T14:35:30.995548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset_classification = img_subset_classification.merge(df_count, on=\"seq_id\")\nimg_subset_classification","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:30.997304Z","iopub.execute_input":"2022-07-21T14:35:30.998057Z","iopub.status.idle":"2022-07-21T14:35:31.038124Z","shell.execute_reply.started":"2022-07-21T14:35:30.998023Z","shell.execute_reply":"2022-07-21T14:35:31.037086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset_classification['count_diff'] = img_subset_classification['bbox_count'] - img_subset_classification['count']\nimg_subset_classification","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.039257Z","iopub.execute_input":"2022-07-21T14:35:31.039930Z","iopub.status.idle":"2022-07-21T14:35:31.074770Z","shell.execute_reply.started":"2022-07-21T14:35:31.039891Z","shell.execute_reply":"2022-07-21T14:35:31.073697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(img_subset_classification['count_diff'])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.076387Z","iopub.execute_input":"2022-07-21T14:35:31.077116Z","iopub.status.idle":"2022-07-21T14:35:31.207678Z","shell.execute_reply.started":"2022-07-21T14:35:31.077072Z","shell.execute_reply":"2022-07-21T14:35:31.206704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**remove images with positive differences from sequence count (since in one image cannot be more animals than in the whole sequence) and remove images with high negative sequence (since it is unlikely to have many animals in one image and almost no animal in another one within the same sequence) in order to reduce error propagation**","metadata":{}},{"cell_type":"code","source":"img_subset_classification = img_subset_classification.loc[(img_subset_classification['count_diff'] <= 0) & (img_subset_classification['count_diff'] >= -5)]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.209230Z","iopub.execute_input":"2022-07-21T14:35:31.209953Z","iopub.status.idle":"2022-07-21T14:35:31.220150Z","shell.execute_reply.started":"2022-07-21T14:35:31.209897Z","shell.execute_reply":"2022-07-21T14:35:31.219381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(img_subset_classification['count_diff'])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.224124Z","iopub.execute_input":"2022-07-21T14:35:31.224558Z","iopub.status.idle":"2022-07-21T14:35:31.405547Z","shell.execute_reply.started":"2022-07-21T14:35:31.224529Z","shell.execute_reply":"2022-07-21T14:35:31.404816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset_classification = img_subset_classification.drop([\"seq_id\",\"category_id\",\"max_detection_conf\",\"detections\",\"count\",\"count_diff\"],axis=1)\nimg_subset_classification","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.406890Z","iopub.execute_input":"2022-07-21T14:35:31.407450Z","iopub.status.idle":"2022-07-21T14:35:31.420912Z","shell.execute_reply.started":"2022-07-21T14:35:31.407418Z","shell.execute_reply":"2022-07-21T14:35:31.420154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**plot annotation distribution**","metadata":{}},{"cell_type":"code","source":"plt.hist(img_subset_classification['bbox_count'])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.422430Z","iopub.execute_input":"2022-07-21T14:35:31.423045Z","iopub.status.idle":"2022-07-21T14:35:31.613085Z","shell.execute_reply.started":"2022-07-21T14:35:31.423012Z","shell.execute_reply":"2022-07-21T14:35:31.611991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Subsample test dataset for submission**","metadata":{}},{"cell_type":"code","source":"test_seq_count = df_count.sample(frac=1,random_state=42).head(100)\ntest_seq_count = test_seq_count.merge(df_countedimageseq, on=\"seq_id\")\ntest_seq_count","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.629337Z","iopub.execute_input":"2022-07-21T14:35:31.629823Z","iopub.status.idle":"2022-07-21T14:35:31.651968Z","shell.execute_reply.started":"2022-07-21T14:35:31.629793Z","shell.execute_reply":"2022-07-21T14:35:31.651198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finaltestimages = pd.DataFrame(test_meta['images'])\ndf_finaltestimages = df_finaltestimages.drop([\"seq_num_frames\",\"datetime\",\"width\",\"width\",\"height\",\"id\",\"sub_location\",\"location\"],axis=1)\nprint(len(df_finaltestimages.index))\ndf_finalseq = df_finaltestimages[\"seq_id\"].drop_duplicates().head(200)\ndf_finaltestimages = df_finaltestimages.merge(df_finalseq, on=\"seq_id\")\ndf_finaltestimages","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.653145Z","iopub.execute_input":"2022-07-21T14:35:31.653631Z","iopub.status.idle":"2022-07-21T14:35:31.906628Z","shell.execute_reply.started":"2022-07-21T14:35:31.653601Z","shell.execute_reply":"2022-07-21T14:35:31.905563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_subset_classification.to_csv(r'./annotation_classification.csv', index = False)\ntest_seq_count.to_csv(r'./test_seq_count_classification.csv', index = False)\ndf_finaltestimages.to_csv(r'./finaleval.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T14:35:31.909519Z","iopub.execute_input":"2022-07-21T14:35:31.909910Z","iopub.status.idle":"2022-07-21T14:35:31.944163Z","shell.execute_reply.started":"2022-07-21T14:35:31.909878Z","shell.execute_reply":"2022-07-21T14:35:31.943215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!mkdir train\n#!mkdir test\n#!mkdir finaleval","metadata":{"execution":{"iopub.status.busy":"2022-07-21T07:12:37.450610Z","iopub.execute_input":"2022-07-21T07:12:37.451470Z","iopub.status.idle":"2022-07-21T07:12:39.828126Z","shell.execute_reply.started":"2022-07-21T07:12:37.451424Z","shell.execute_reply":"2022-07-21T07:12:39.826715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#move train images to train\n#for index, row in img_subset_classification.iterrows():\n#    file_location = \"../input/iwildcam2022-fgvc9/train/train/\" + row[\"file_name\"]\n#    file_target = \"./train/\" + row[\"file_name\"]\n#    !cp \"$file_location\" \"$file_target\"\n\n#move test images to test\n#for index, row in test_seq_count.iterrows():\n#    file_location = \"../input/iwildcam2022-fgvc9/train/train/\" + row[\"file_name\"]\n#    file_target = \"./test/\" + row[\"file_name\"]\n#    !cp \"$file_location\" \"$file_target\"\n\n\n#move finaleval images to finaleval\n#for index, row in df_finaltestimages.iterrows():\n#    file_location = \"../input/iwildcam2022-fgvc9/test/test/\" + row[\"file_name\"]\n#    file_target = \"./finaleval/\" + row[\"file_name\"]\n#    !cp \"$file_location\" \"$file_target\"","metadata":{"execution":{"iopub.status.busy":"2022-07-21T07:23:08.885014Z","iopub.execute_input":"2022-07-21T07:23:08.885426Z","iopub.status.idle":"2022-07-21T08:50:47.104211Z","shell.execute_reply.started":"2022-07-21T07:23:08.885388Z","shell.execute_reply":"2022-07-21T08:50:47.101797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!zip -r ./train.zip ./train\n#!zip -r ./test.zip ./test\n#!zip -r ./finaleval.zip ./finaleval","metadata":{"execution":{"iopub.status.busy":"2022-07-21T09:40:18.376583Z","iopub.execute_input":"2022-07-21T09:40:18.377844Z","iopub.status.idle":"2022-07-21T09:40:18.396661Z","shell.execute_reply.started":"2022-07-21T09:40:18.377730Z","shell.execute_reply":"2022-07-21T09:40:18.395637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from IPython.display import FileLink\n#FileLink(\"./train.zip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:57:28.872276Z","iopub.execute_input":"2022-07-21T08:57:28.873453Z","iopub.status.idle":"2022-07-21T08:57:28.882248Z","shell.execute_reply.started":"2022-07-21T08:57:28.873401Z","shell.execute_reply":"2022-07-21T08:57:28.880859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FileLink(\"./test.zip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:58:32.257199Z","iopub.execute_input":"2022-07-21T08:58:32.257772Z","iopub.status.idle":"2022-07-21T08:58:32.267967Z","shell.execute_reply.started":"2022-07-21T08:58:32.257734Z","shell.execute_reply":"2022-07-21T08:58:32.266912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FileLink(\"./finaleval.zip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:58:32.761424Z","iopub.execute_input":"2022-07-21T08:58:32.762239Z","iopub.status.idle":"2022-07-21T08:58:32.770180Z","shell.execute_reply.started":"2022-07-21T08:58:32.762170Z","shell.execute_reply":"2022-07-21T08:58:32.769097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FileLink(\"./annotation_classification.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:58:38.987823Z","iopub.execute_input":"2022-07-21T08:58:38.988858Z","iopub.status.idle":"2022-07-21T08:58:38.996774Z","shell.execute_reply.started":"2022-07-21T08:58:38.988809Z","shell.execute_reply":"2022-07-21T08:58:38.995756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FileLink(\"./test_seq_count_classification.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:58:39.575552Z","iopub.execute_input":"2022-07-21T08:58:39.576457Z","iopub.status.idle":"2022-07-21T08:58:39.583413Z","shell.execute_reply.started":"2022-07-21T08:58:39.576415Z","shell.execute_reply":"2022-07-21T08:58:39.582366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FileLink(\"./finaleval.csv\"","metadata":{},"execution_count":null,"outputs":[]}]}