{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Content:\n\n1 Initializing and loading data\n\n2 Routines\n\n2.1. Function show_imgs_as_tiles\n\n2.2. Subsample dataframe picking top n records for each group    \n\n3 Quick analysis\n\n4 View some samples\n\n4.1. Sample 10 unique individuals from each specie\n\n4.2. Sample same individual different images","metadata":{}},{"cell_type":"markdown","source":"# 1. Initializing and loading data","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ntrain_path = '/kaggle/input/happy-whale-and-dolphin/train_images/'\ntrain_files = []\nfor dirname, _, filenames in os.walk(train_path):\n    for filename in filenames:\n        train_files.append(os.path.join(dirname, filename))\ntrain = pd.read_csv('../input/happy-whale-and-dolphin/train.csv')\n \ntest_path = '/kaggle/input/happy-whale-and-dolphin/test_images/'\ntest_files = []\nfor dirname, _, filenames in os.walk(test_path):\n    for filename in filenames:\n        test_files.append(os.path.join(dirname, filename))\ntest = pd.read_csv('../input/happy-whale-and-dolphin/sample_submission.csv')    \n    \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-06T12:22:33.136691Z","iopub.execute_input":"2022-02-06T12:22:33.137702Z","iopub.status.idle":"2022-02-06T12:22:48.441527Z","shell.execute_reply.started":"2022-02-06T12:22:33.137625Z","shell.execute_reply":"2022-02-06T12:22:48.440571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Some quick check if the input is OK:\")\n\nprint(f\"Training set {len(train_files)} images\") \nprint(f\"Training data: {len(train)} records\")\nprint(f\"Unique species: {train['species'].nunique()}\")\nprint(f\"Unique individuals: {train['individual_id'].nunique()}\")\n\nprint(f\"Test set {len(test_files)} images\") \nprint(f\"Test data: {len(test)} in submit sample\")","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:47:01.669418Z","iopub.execute_input":"2022-02-06T12:47:01.669751Z","iopub.status.idle":"2022-02-06T12:47:01.686741Z","shell.execute_reply.started":"2022-02-06T12:47:01.669719Z","shell.execute_reply":"2022-02-06T12:47:01.685861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"predict1\"] = test[\"predictions\"].str.split(\" \", 1).str[0]\ntest[\"predict2\"] = test[\"predictions\"].str.split(\" \", 2).str[1]\ntest[\"predict3\"] = test[\"predictions\"].str.split(\" \", 3).str[2]\ntest[\"predict4\"] = test[\"predictions\"].str.split(\" \", 4).str[3]\ntest[\"predict5\"] = test[\"predictions\"].str.split(\" \", 5).str[4]\ntest.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:29:48.749932Z","iopub.execute_input":"2022-02-06T12:29:48.750203Z","iopub.status.idle":"2022-02-06T12:29:48.964412Z","shell.execute_reply.started":"2022-02-06T12:29:48.750176Z","shell.execute_reply":"2022-02-06T12:29:48.963578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Routines","metadata":{}},{"cell_type":"markdown","source":"## 2.1 Function show_imgs_as_tiles","metadata":{}},{"cell_type":"code","source":"import PIL\nimport matplotlib.pyplot as plt\nimport os\n\ndef show_imgs_as_tiles( w, h, imgs, labels=None, tile_width=200, tile_height=200, \\\n                       path='/kaggle/input/happy-whale-and-dolphin/train_images/'): \n    \"\"\" display w, h tiles with images and labels \n    \"\"\"\n    # this function uses the open, resize and array functions we have seen before\n    load_img = lambda filename: np.array(PIL.Image.open(f\"{filename}\").resize((tile_width, tile_height)))\n    \n    _, axes_list = plt.subplots(h, w, figsize=(2*w, 2*h)) # define a grid of (w, h)\n    \n    i = 0 \n    for axes in axes_list:\n        for ax in axes:\n            if i<len(imgs):\n                img = os.path.join( path, imgs[i])\n                ax.axis('off')\n                ax.imshow(load_img(img)) # load and show\n                if len(labels[i])>18:\n                    ax.set_title(labels[i].replace('.jpg','')[-18:])\n                else:\n                    ax.set_title(labels[i].replace('.jpg',''))\n            else:\n                ax.axis('off')\n            i+=1\n\n# quick test                 \nshow_imgs_as_tiles(w=4, h=2, imgs= train['image'], labels=imgs, tile_width=200, tile_height=200)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:10:51.645402Z","iopub.execute_input":"2022-02-06T12:10:51.645682Z","iopub.status.idle":"2022-02-06T12:10:52.365167Z","shell.execute_reply.started":"2022-02-06T12:10:51.645638Z","shell.execute_reply":"2022-02-06T12:10:52.364078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2 Subsample dataframe picking top n records for each group","metadata":{}},{"cell_type":"code","source":"def sub_sample_by(df=pd.DataFrame(), by='species', groups=top_species, n_samples=5):\n    \"\"\" Subsample a df by picking top n records for each group\n    \"\"\"\n    sample = []\n    for s in groups:\n        ind = df[ df[by] == s  ].iloc[:n_samples]\n        sample.append(ind)\n    return pd.concat(sample)\n\nsub_sample_by( df=train, by='species', groups=train.species.unique(), n_samples=3 )","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:09:23.064384Z","iopub.execute_input":"2022-02-06T12:09:23.065267Z","iopub.status.idle":"2022-02-06T12:09:23.197362Z","shell.execute_reply.started":"2022-02-06T12:09:23.065204Z","shell.execute_reply":"2022-02-06T12:09:23.196478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Quick analysis","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:09:27.632327Z","iopub.execute_input":"2022-02-06T12:09:27.632584Z","iopub.status.idle":"2022-02-06T12:09:27.642507Z","shell.execute_reply.started":"2022-02-06T12:09:27.632556Z","shell.execute_reply":"2022-02-06T12:09:27.641582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"species = train.groupby('species')['image'].count().reset_index(name='count_specie').sort_values(['count_specie'], ascending=False)\ntrain = pd.merge(train, species, left_on='species', right_on='species')\nspecies.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:09:30.406586Z","iopub.execute_input":"2022-02-06T12:09:30.407253Z","iopub.status.idle":"2022-02-06T12:09:30.436224Z","shell.execute_reply.started":"2022-02-06T12:09:30.407207Z","shell.execute_reply":"2022-02-06T12:09:30.435394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"individuals = train.groupby('individual_id')['image'].count().reset_index(name='count_individ').sort_values(['count_individ'], ascending=False)\ntrain = pd.merge(train, individuals, left_on='individual_id', right_on='individual_id')\nindividuals.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:09:33.696943Z","iopub.execute_input":"2022-02-06T12:09:33.697205Z","iopub.status.idle":"2022-02-06T12:09:33.753441Z","shell.execute_reply.started":"2022-02-06T12:09:33.697175Z","shell.execute_reply":"2022-02-06T12:09:33.752555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:09:37.739768Z","iopub.execute_input":"2022-02-06T12:09:37.740105Z","iopub.status.idle":"2022-02-06T12:09:37.752948Z","shell.execute_reply.started":"2022-02-06T12:09:37.740070Z","shell.execute_reply":"2022-02-06T12:09:37.751845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. View some samples","metadata":{}},{"cell_type":"markdown","source":"## 4.1. Sample 10 unique individuals from each specie","metadata":{}},{"cell_type":"code","source":"n, m = 30, 10  # n species x m individuals \ntop_species = train.drop_duplicates(['individual_id'], keep='first')\\\n                    .groupby(by='species')\\\n                    .sample(n=m)\\\n                    .sort_values(by=['count_specie'], ascending=[False])\ntop_species.head(25)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:09:54.952770Z","iopub.execute_input":"2022-02-06T12:09:54.953054Z","iopub.status.idle":"2022-02-06T12:09:54.990390Z","shell.execute_reply.started":"2022-02-06T12:09:54.953018Z","shell.execute_reply":"2022-02-06T12:09:54.989539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = list(top_species.image)\nlabels = list(top_species.species)\nshow_imgs_as_tiles(w=m, h=n, imgs=imgs, labels=labels,tile_width=200, tile_height=200, path=train_path)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T12:11:27.079065Z","iopub.execute_input":"2022-02-06T12:11:27.079329Z","iopub.status.idle":"2022-02-06T12:12:10.650276Z","shell.execute_reply.started":"2022-02-06T12:11:27.079300Z","shell.execute_reply":"2022-02-06T12:12:10.649512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.2. Sample same individual different images","metadata":{}},{"cell_type":"code","source":"n, m = 30, 9  # n species x m individuals \nsame = train.sort_values(by=['count_specie','count_individ', 'individual_id'], ascending=[False,False,True])\nsame = sub_sample_by( df=same, by='species', groups=same.species.unique(), n_samples=m )\nsame.head(25)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T11:14:13.128854Z","iopub.execute_input":"2022-02-06T11:14:13.129100Z","iopub.status.idle":"2022-02-06T11:14:13.288320Z","shell.execute_reply.started":"2022-02-06T11:14:13.129070Z","shell.execute_reply":"2022-02-06T11:14:13.287706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = list(same.image)\nlabels = [ x[:7] + y[-7:] for x, y in zip( list(same.species) , list(same.individual_id) )]\nshow_imgs_as_tiles(w=m, h=n, imgs=imgs, labels=labels,tile_width=200, tile_height=200, path=train_path)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T11:20:52.547989Z","iopub.execute_input":"2022-02-06T11:20:52.548290Z","iopub.status.idle":"2022-02-06T11:21:33.458904Z","shell.execute_reply.started":"2022-02-06T11:20:52.548250Z","shell.execute_reply":"2022-02-06T11:21:33.456982Z"},"trusted":true},"execution_count":null,"outputs":[]}]}