{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-03T08:30:19.909135Z","iopub.execute_input":"2021-12-03T08:30:19.909524Z","iopub.status.idle":"2021-12-03T08:30:25.875788Z","shell.execute_reply.started":"2021-12-03T08:30:19.909372Z","shell.execute_reply":"2021-12-03T08:30:25.874136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#导入和模块","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport os\nimport pathlib\nimport PIL\nfrom pathlib import Path\nfrom PIL import Image, ImageDraw\nfrom math import sqrt\nimport ast\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nsns.set()\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:32:34.698151Z","iopub.execute_input":"2021-12-03T08:32:34.698425Z","iopub.status.idle":"2021-12-03T08:32:35.115971Z","shell.execute_reply.started":"2021-12-03T08:32:34.698399Z","shell.execute_reply":"2021-12-03T08:32:35.115115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##训练","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/tensorflow-great-barrier-reef/train.csv\")\ntest = pd.read_csv(\"../input/tensorflow-great-barrier-reef/test.csv\")\nsub = pd.read_csv(\"../input/tensorflow-great-barrier-reef/example_sample_submission.csv\")\n\npath = Path('../input/tensorflow-great-barrier-reef/train_images')\nfilepaths = list(path.glob(r'**/*.jpg'))","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:32:38.240724Z","iopub.execute_input":"2021-12-03T08:32:38.241274Z","iopub.status.idle":"2021-12-03T08:32:41.134005Z","shell.execute_reply.started":"2021-12-03T08:32:38.241234Z","shell.execute_reply":"2021-12-03T08:32:41.133137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the train test lengths\nprint(\"Number of training samples: \", len(train))\nprint(\"Number of testing samples: \", len(test))","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:32:59.945366Z","iopub.execute_input":"2021-12-03T08:32:59.945863Z","iopub.status.idle":"2021-12-03T08:32:59.951811Z","shell.execute_reply.started":"2021-12-03T08:32:59.945827Z","shell.execute_reply":"2021-12-03T08:32:59.951139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(150)","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:33:08.960737Z","iopub.execute_input":"2021-12-03T08:33:08.961137Z","iopub.status.idle":"2021-12-03T08:33:08.984498Z","shell.execute_reply.started":"2021-12-03T08:33:08.961105Z","shell.execute_reply":"2021-12-03T08:33:08.983946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#看起来很多镜框都没有我们的海星","metadata":{}},{"cell_type":"code","source":"# lets see how many frames with no starfishes\ntrain_clean = train.loc[train[\"annotations\"] != \"[]\"]\nprint(f\"No starfishes in {len(train)-len(train_clean)} samples.\")\nprint(f\"The clean train set has {len(train_clean)} images for us to work with.\")","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:34:12.018187Z","iopub.execute_input":"2021-12-03T08:34:12.018591Z","iopub.status.idle":"2021-12-03T08:34:12.031734Z","shell.execute_reply.started":"2021-12-03T08:34:12.018560Z","shell.execute_reply":"2021-12-03T08:34:12.030922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_clean.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:34:21.439806Z","iopub.execute_input":"2021-12-03T08:34:21.440626Z","iopub.status.idle":"2021-12-03T08:34:21.451687Z","shell.execute_reply.started":"2021-12-03T08:34:21.440578Z","shell.execute_reply":"2021-12-03T08:34:21.450919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##序列分布","metadata":{}},{"cell_type":"code","source":"# Checking out the number of sequences\nlen(train_clean.sequence.value_counts())","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:34:45.934038Z","iopub.execute_input":"2021-12-03T08:34:45.934566Z","iopub.status.idle":"2021-12-03T08:34:45.945082Z","shell.execute_reply.started":"2021-12-03T08:34:45.934519Z","shell.execute_reply":"2021-12-03T08:34:45.944189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rows per each sequence\nprint(\"Sequence Samples\")\nprint(train_clean.sequence.value_counts())","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:34:54.082399Z","iopub.execute_input":"2021-12-03T08:34:54.082816Z","iopub.status.idle":"2021-12-03T08:34:54.089303Z","shell.execute_reply.started":"2021-12-03T08:34:54.082777Z","shell.execute_reply":"2021-12-03T08:34:54.088356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq_df = train_clean.sequence.value_counts().to_frame()\nplt.figure(figsize=(16, 9))\nsns.barplot(x=seq_df.index, y=list(seq_df.sequence), palette=\"Greens_d\")\nplt.title(\"Distribution of Sequences\")\nplt.xlabel(\"Sequence Id\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:35:02.416498Z","iopub.execute_input":"2021-12-03T08:35:02.417069Z","iopub.status.idle":"2021-12-03T08:35:02.796050Z","shell.execute_reply.started":"2021-12-03T08:35:02.417032Z","shell.execute_reply":"2021-12-03T08:35:02.795404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##海星的数量","metadata":{}},{"cell_type":"code","source":"num_boxes = []\nannotations_clean = []\nfor elem in train_clean.annotations:\n    ann = ast.literal_eval(elem)\n    num_boxes.append(len(ann))\n    annotations_clean.append(ann)","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:35:40.393551Z","iopub.execute_input":"2021-12-03T08:35:40.393978Z","iopub.status.idle":"2021-12-03T08:35:40.595812Z","shell.execute_reply.started":"2021-12-03T08:35:40.393919Z","shell.execute_reply":"2021-12-03T08:35:40.595132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adding num boxes per row and changing the annotations column to a proper python parseable list of dictionaries\ntrain_clean[\"num_boxes\"] = num_boxes\ntrain_clean[\"annotations\"] = annotations_clean","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:35:50.982874Z","iopub.execute_input":"2021-12-03T08:35:50.983307Z","iopub.status.idle":"2021-12-03T08:35:50.991222Z","shell.execute_reply.started":"2021-12-03T08:35:50.983276Z","shell.execute_reply":"2021-12-03T08:35:50.990389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_clean.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:35:58.583752Z","iopub.execute_input":"2021-12-03T08:35:58.584381Z","iopub.status.idle":"2021-12-03T08:35:58.597521Z","shell.execute_reply.started":"2021-12-03T08:35:58.584337Z","shell.execute_reply":"2021-12-03T08:35:58.596978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"#box Frequency\")\nprint(train_clean.num_boxes.value_counts())","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:36:06.527755Z","iopub.execute_input":"2021-12-03T08:36:06.528261Z","iopub.status.idle":"2021-12-03T08:36:06.534527Z","shell.execute_reply.started":"2021-12-03T08:36:06.528219Z","shell.execute_reply":"2021-12-03T08:36:06.533646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of bounding boxes in the clean train datasets\nprint(f\"Number of Bounding Boxes in the dataset: {train_clean.num_boxes.sum()}\")","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:36:16.423153Z","iopub.execute_input":"2021-12-03T08:36:16.423634Z","iopub.status.idle":"2021-12-03T08:36:16.427962Z","shell.execute_reply.started":"2021-12-03T08:36:16.423598Z","shell.execute_reply":"2021-12-03T08:36:16.427391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##边界框数的分布","metadata":{}},{"cell_type":"code","source":"box_count = train_clean.num_boxes.value_counts().to_frame()","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:36:46.999657Z","iopub.execute_input":"2021-12-03T08:36:47.000224Z","iopub.status.idle":"2021-12-03T08:36:47.005666Z","shell.execute_reply.started":"2021-12-03T08:36:47.000175Z","shell.execute_reply":"2021-12-03T08:36:47.005101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 9))\nsns.barplot(x=box_count.index, y=list(box_count.num_boxes), palette=\"Greens_d\")\nplt.title(\"Distribution of Num_boxes\")\nplt.xlabel(\"# of Boxes\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:37:59.505354Z","iopub.execute_input":"2021-12-03T08:37:59.505918Z","iopub.status.idle":"2021-12-03T08:37:59.845148Z","shell.execute_reply.started":"2021-12-03T08:37:59.505860Z","shell.execute_reply":"2021-12-03T08:37:59.844324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"看起来57%的数据点只有一个边界框，其次是19.1%的数据点有两个边界框","metadata":{}},{"cell_type":"markdown","source":"##观察海星","metadata":{}},{"cell_type":"code","source":"#structure of a annotation\nlist(train_clean[\"annotations\"])[0]","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:38:48.806993Z","iopub.execute_input":"2021-12-03T08:38:48.807547Z","iopub.status.idle":"2021-12-03T08:38:48.814216Z","shell.execute_reply.started":"2021-12-03T08:38:48.807510Z","shell.execute_reply":"2021-12-03T08:38:48.813420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# generating paths for input images\nsrc = '../input/tensorflow-great-barrier-reef/train_images'\npaths = []\nfor row in train_clean.image_id:\n    vid_num = row.split('-')[0]\n    img_num = row.split('-')[1]\n    paths.append(os.path.join(src,f'video_{vid_num}',img_num+'.jpg'))","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:38:57.441948Z","iopub.execute_input":"2021-12-03T08:38:57.442710Z","iopub.status.idle":"2021-12-03T08:38:57.459077Z","shell.execute_reply.started":"2021-12-03T08:38:57.442660Z","shell.execute_reply":"2021-12-03T08:38:57.458188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_clean['paths'] = paths","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:39:05.936260Z","iopub.execute_input":"2021-12-03T08:39:05.936580Z","iopub.status.idle":"2021-12-03T08:39:05.940940Z","shell.execute_reply.started":"2021-12-03T08:39:05.936545Z","shell.execute_reply":"2021-12-03T08:39:05.940389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# classic way of iterating through and drawing the bounding boxes on an image\ndef vis_boxes(img_path, bboxes):\n    coords = []\n    for box in bboxes:\n        x1 = box['x']\n        y1 = box['y']\n        x2 = x1 + box['width']\n        y2 = y1 + box['height']\n        coords.append([x1, y1, x2, y2])\n        \n    img = Image.open(img_path)\n    img1 = img.copy()\n    draw = ImageDraw.Draw(img1)\n    for elem in coords:\n        draw.rectangle(elem, outline='red', width=7)\n    \n    return img1","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:39:13.914390Z","iopub.execute_input":"2021-12-03T08:39:13.914638Z","iopub.status.idle":"2021-12-03T08:39:13.921368Z","shell.execute_reply.started":"2021-12-03T08:39:13.914612Z","shell.execute_reply":"2021-12-03T08:39:13.920617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_clean.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:39:21.551305Z","iopub.execute_input":"2021-12-03T08:39:21.551999Z","iopub.status.idle":"2021-12-03T08:39:21.565291Z","shell.execute_reply.started":"2021-12-03T08:39:21.551960Z","shell.execute_reply":"2021-12-03T08:39:21.564624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##具有最大边界框的序列","metadata":{}},{"cell_type":"code","source":"# number of bounding boxes per each sequence\ntrain_clean.groupby('sequence').num_boxes.sum().to_frame()","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:39:35.280641Z","iopub.execute_input":"2021-12-03T08:39:35.281208Z","iopub.status.idle":"2021-12-03T08:39:35.295905Z","shell.execute_reply.started":"2021-12-03T08:39:35.281168Z","shell.execute_reply":"2021-12-03T08:39:35.295347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##看一些例子","metadata":{}},{"cell_type":"code","source":"# lets plot a few\n# some inspiration from https://www.kaggle.com/sjyangkevin/eda-bounding-box-analysis-annotated-videos\n\nplt.figure(figsize=(16, 9))\nn_images = 9\ncount = 0\nr,c = int(sqrt(n_images)), int(sqrt(n_images))\ntrain_plot = train_clean.sample(n = n_images)\n\nfor _, row in train_plot.iterrows():\n    img_path = row['paths']\n    bboxes = row['annotations']\n    plt.subplot(r, c, count + 1)\n    img_out = vis_boxes(img_path, bboxes)\n    plt.imshow(img_out)\n    count+=1\n\nplt.show()\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2021-12-03T08:40:11.983608Z","iopub.execute_input":"2021-12-03T08:40:11.984245Z","iopub.status.idle":"2021-12-03T08:40:14.759672Z","shell.execute_reply.started":"2021-12-03T08:40:11.984205Z","shell.execute_reply":"2021-12-03T08:40:14.758988Z"},"trusted":true},"execution_count":null,"outputs":[]}]}