{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# GeoLifeCLEF2022 - Exploratory Data Analysis\n\nOn-Going EDA","metadata":{}},{"cell_type":"code","source":"%pylab inline --no-import-all\n\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport umap","metadata":{"execution":{"iopub.status.busy":"2022-03-19T16:30:48.823107Z","iopub.execute_input":"2022-03-19T16:30:48.823463Z","iopub.status.idle":"2022-03-19T16:30:48.835044Z","shell.execute_reply.started":"2022-03-19T16:30:48.823431Z","shell.execute_reply":"2022-03-19T16:30:48.834112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf GLC\n!git clone https://github.com/maximiliense/GLC","metadata":{"ExecuteTime":{"end_time":"2022-02-15T14:44:43.556762Z","start_time":"2022-02-15T14:44:42.730071Z"},"execution":{"iopub.status.busy":"2022-03-14T12:18:37.496901Z","iopub.execute_input":"2022-03-14T12:18:37.497556Z","iopub.status.idle":"2022-03-14T12:18:40.545705Z","shell.execute_reply.started":"2022-03-14T12:18:37.497429Z","shell.execute_reply":"2022-03-14T12:18:40.5446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = Path(\"../input/geolifeclef-2022-lifeclef-2022-fgvc9/\")","metadata":{"execution":{"iopub.status.busy":"2022-03-19T16:31:17.964882Z","iopub.execute_input":"2022-03-19T16:31:17.965190Z","iopub.status.idle":"2022-03-19T16:31:17.969827Z","shell.execute_reply.started":"2022-03-19T16:31:17.965158Z","shell.execute_reply":"2022-03-19T16:31:17.968792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tools","metadata":{}},{"cell_type":"code","source":"from GLC.plotting import plot_map\n\ndef plot_observations_distribution(ax, df_obs, df_obs_test=None, **kwargs):\n    default_kwargs = {\n        \"zorder\": 1,\n        \"alpha\": 0.1,\n        \"s\": 0.5\n    }\n    default_kwargs.update(kwargs)\n    kwargs = default_kwargs\n    \n    ax.scatter(df_obs.longitude, df_obs.latitude, color=\"blue\", **kwargs)\n    \n    if df_obs_test is not None:\n        ax.scatter(df_obs_test.longitude, df_obs_test.latitude, color=\"red\", **kwargs)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:18:40.549661Z","iopub.execute_input":"2022-03-14T12:18:40.549964Z","iopub.status.idle":"2022-03-14T12:18:40.561105Z","shell.execute_reply.started":"2022-03-14T12:18:40.549931Z","shell.execute_reply":"2022-03-14T12:18:40.560405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observations","metadata":{}},{"cell_type":"code","source":"df_obs_fr = pd.read_csv(DATA_PATH / \"observations\" / \"observations_fr_train.csv\", sep=\";\", index_col=\"observation_id\")\ndf_obs_us = pd.read_csv(DATA_PATH / \"observations\" / \"observations_us_train.csv\", sep=\";\", index_col=\"observation_id\")\ndf_obs_fr_test = pd.read_csv(DATA_PATH / \"observations\" / \"observations_fr_test.csv\", sep=\";\", index_col=\"observation_id\")\ndf_obs_us_test = pd.read_csv(DATA_PATH / \"observations\" / \"observations_us_test.csv\", sep=\";\", index_col=\"observation_id\")\n\ndf_obs = pd.concat((df_obs_fr, df_obs_us))\ndf_obs_test = pd.concat((df_obs_fr_test, df_obs_us_test))\n\nprint(f\"Number of observations for training: {len(df_obs)}\")\nprint(f\"Number of observations for testing: {len(df_obs_test)}\")\n\ndf_obs.head()","metadata":{"ExecuteTime":{"end_time":"2022-02-15T14:44:45.101516Z","start_time":"2022-02-15T14:44:44.129115Z"},"execution":{"iopub.status.busy":"2022-03-14T12:18:40.562871Z","iopub.execute_input":"2022-03-14T12:18:40.563284Z","iopub.status.idle":"2022-03-14T12:18:42.951708Z","shell.execute_reply.started":"2022-03-14T12:18:40.563249Z","shell.execute_reply":"2022-03-14T12:18:42.950686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target Distribution","metadata":{}},{"cell_type":"code","source":"freq = df_obs.species_id.value_counts()\nprop_filter = 0.95\ntop_k = 30\n\nprint(f'{len(freq)} unique species')\n\nprint('top 5 species:')\nprint(freq.head())\n\nprint(f'Top {top_k} proportion of total: {freq.cumsum().values[top_k-1]/len(df_obs):.2%}')\nprint(f'{(freq.cumsum()/len(df_obs)<prop_filter).mean():.2%} targets cumulates to {prop_filter:.2%} of observations')\n\nplt.plot(freq.cumsum().values/len(df_obs.species_id));\nplt.axhline(y=prop_filter, color='r', linestyle='-');\nplt.axvline(x=30, color='k', linestyle='-');","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:18:42.954272Z","iopub.execute_input":"2022-03-14T12:18:42.95461Z","iopub.status.idle":"2022-03-14T12:18:43.261935Z","shell.execute_reply.started":"2022-03-14T12:18:42.954565Z","shell.execute_reply":"2022-03-14T12:18:43.261274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target Types","metadata":{}},{"cell_type":"code","source":"species_details = pd.read_csv(DATA_PATH / \"metadata\" / \"species_details.csv\", sep=\";\")\nspecies_details['count'] = species_details.species_id.map(freq)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:18:43.262921Z","iopub.execute_input":"2022-03-14T12:18:43.263533Z","iopub.status.idle":"2022-03-14T12:18:43.317828Z","shell.execute_reply.started":"2022-03-14T12:18:43.263499Z","shell.execute_reply":"2022-03-14T12:18:43.316648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"species_details.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:18:43.319245Z","iopub.execute_input":"2022-03-14T12:18:43.320618Z","iopub.status.idle":"2022-03-14T12:18:43.333336Z","shell.execute_reply.started":"2022-03-14T12:18:43.320577Z","shell.execute_reply":"2022-03-14T12:18:43.332484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Kingdoms","metadata":{}},{"cell_type":"code","source":"gb = species_details.groupby('GBIF_kingdom_name')['count'].sum()\nplt.pie(gb.values.flatten().astype('int'),labels = gb.index);","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:18:43.334694Z","iopub.execute_input":"2022-03-14T12:18:43.3351Z","iopub.status.idle":"2022-03-14T12:18:43.582446Z","shell.execute_reply.started":"2022-03-14T12:18:43.335051Z","shell.execute_reply":"2022-03-14T12:18:43.581421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Families","metadata":{}},{"cell_type":"code","source":"gb = species_details.groupby('GBIF_family_name')['count'].sum()\n\nfreq = gb.sort_values(ascending=False)\nprop_filter = 0.95\ntop_k = 30\n\nprint(f'{len(freq)} unique families')\nprint('top 5 families:')\nprint(freq.head())\n\nprint(f'Top {top_k} families proportion of total: {freq.cumsum().values[top_k-1]/len(df_obs):.2%}')\nprint(f'{(freq.cumsum()/len(df_obs)<prop_filter).mean():.2%} families cumulates to {prop_filter:.2%} of observations')\n\nplt.plot(freq.cumsum().values/len(df_obs.species_id));\nplt.axhline(y=prop_filter, color='r', linestyle='-');\nplt.axvline(x=top_k, color='k', linestyle='-');","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:18:43.584377Z","iopub.execute_input":"2022-03-14T12:18:43.584852Z","iopub.status.idle":"2022-03-14T12:18:43.82707Z","shell.execute_reply.started":"2022-03-14T12:18:43.584801Z","shell.execute_reply":"2022-03-14T12:18:43.826139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lots of plants at the top.","metadata":{"execution":{"iopub.status.busy":"2022-03-13T16:39:14.438889Z","iopub.execute_input":"2022-03-13T16:39:14.439163Z","iopub.status.idle":"2022-03-13T16:39:14.444054Z","shell.execute_reply.started":"2022-03-13T16:39:14.439135Z","shell.execute_reply":"2022-03-13T16:39:14.443237Z"}}},{"cell_type":"markdown","source":"# Unique Species","metadata":{"execution":{"iopub.status.busy":"2022-03-13T15:51:52.599381Z","iopub.execute_input":"2022-03-13T15:51:52.599907Z","iopub.status.idle":"2022-03-13T15:51:52.60485Z","shell.execute_reply.started":"2022-03-13T15:51:52.599862Z","shell.execute_reply":"2022-03-13T15:51:52.60431Z"}}},{"cell_type":"code","source":"specie_id = 5045\n\nfig = plt.figure(figsize=(10, 5.5))\nax = plot_map(region=\"us\")\nplot_observations_distribution(ax, df_obs_us, df_obs_us[df_obs_us.species_id==specie_id ])\nax.set_title(f\"Observations distribution (US) - specie {specie_id}\")","metadata":{"ExecuteTime":{"end_time":"2022-02-15T14:44:56.860052Z","start_time":"2022-02-15T14:44:45.134254Z"},"execution":{"iopub.status.busy":"2022-03-14T12:18:43.828263Z","iopub.execute_input":"2022-03-14T12:18:43.828481Z","iopub.status.idle":"2022-03-14T12:18:51.764824Z","shell.execute_reply.started":"2022-03-14T12:18:43.828455Z","shell.execute_reply":"2022-03-14T12:18:51.764001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The specie don't appear in France... So it seems we have important geo correlation.","metadata":{}},{"cell_type":"markdown","source":"# patch data\n\nVoluminous data only accessible trough specific function.\nMight need some work befor being usable. ","metadata":{}},{"cell_type":"code","source":"from GLC.data_loading.common import load_patch\nfrom GLC.plotting import visualize_observation_patch\n\npatch = load_patch(10171444, DATA_PATH)\nprint(\"Number of data sources: {}\".format(len(patch)))\nprint(\"Arrays shape: {}\".format([p.shape for p in patch]))\nprint(\"Data types: {}\".format([p.dtype for p in patch]))\n\n\ndf_suggested_landcover_alignment = pd.read_csv(DATA_PATH / \"metadata\" / \"landcover_suggested_alignment.csv\", sep=\";\")\ndf_suggested_landcover_alignment.head()\n\nlandcover_mapping = df_suggested_landcover_alignment[\"suggested_landcover_code\"].values\npatch = load_patch(10171444, DATA_PATH, landcover_mapping=landcover_mapping)\n\nR = patch[0][:,:,0]\nG = patch[0][:,:,1]\nB = patch[0][:,:,2]\nIR = patch[1]\nAltitude = patch[2]\nland_cover = patch[3]","metadata":{"ExecuteTime":{"end_time":"2022-02-15T14:45:00.07101Z","start_time":"2022-02-15T14:45:00.021859Z"},"execution":{"iopub.status.busy":"2022-03-14T12:18:51.767501Z","iopub.execute_input":"2022-03-14T12:18:51.767938Z","iopub.status.idle":"2022-03-14T12:18:52.047803Z","shell.execute_reply.started":"2022-03-14T12:18:51.767904Z","shell.execute_reply":"2022-03-14T12:18:52.046151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Environmental rasters\n","metadata":{}},{"cell_type":"code","source":"df_env = pd.read_csv(DATA_PATH / \"pre-extracted\" / \"environmental_vectors.csv\", sep=\";\")\ndf_env.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-19T16:31:24.427902Z","iopub.execute_input":"2022-03-19T16:31:24.428210Z","iopub.status.idle":"2022-03-19T16:31:37.692894Z","shell.execute_reply.started":"2022-03-19T16:31:24.428177Z","shell.execute_reply":"2022-03-19T16:31:37.691758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_env.hist(bins=100,figsize=(20,20));","metadata":{"execution":{"iopub.status.busy":"2022-03-19T16:31:37.694542Z","iopub.execute_input":"2022-03-19T16:31:37.694820Z","iopub.status.idle":"2022-03-19T16:31:48.460158Z","shell.execute_reply.started":"2022-03-19T16:31:37.694786Z","shell.execute_reply":"2022-03-19T16:31:48.459419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"execution":{"iopub.status.busy":"2022-03-19T16:35:02.891786Z","iopub.execute_input":"2022-03-19T16:35:02.892122Z","iopub.status.idle":"2022-03-19T16:35:07.706614Z","shell.execute_reply.started":"2022-03-19T16:35:02.892091Z","shell.execute_reply":"2022-03-19T16:35:07.705015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = df_env.corr()\n\nfig, ax = plt.subplots(figsize=(16,16))  \n\nsns.heatmap(\n    corr, \n    vmin=-1, vmax=1, center=0,\n    cmap=sns.diverging_palette(20, 220, n=200),\n    square=True, ax=ax\n);\n\nax.set_xticklabels(\n    ax.get_xticklabels(),\n    rotation=45,\n    horizontalalignment='right',\n);","metadata":{"execution":{"iopub.status.busy":"2022-03-19T16:36:36.743043Z","iopub.execute_input":"2022-03-19T16:36:36.743344Z","iopub.status.idle":"2022-03-19T16:36:41.407736Z","shell.execute_reply.started":"2022-03-19T16:36:36.743314Z","shell.execute_reply":"2022-03-19T16:36:41.406524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_env['country'] = df_env.observation_id.astype('str').str[0]\ndf_explo = df_env[df_env.country=='1'].drop(['observation_id','country'],axis=1)\ndf_explo = df_explo.fillna(df_explo.mean())\nmean_explo = df_explo.mean()\nstd_explo = df_explo.std()\ndf_explo = ((df_explo-mean_explo)/std_explo).copy()","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:27:12.470102Z","iopub.execute_input":"2022-03-14T12:27:12.47067Z","iopub.status.idle":"2022-03-14T12:27:16.495795Z","shell.execute_reply.started":"2022-03-14T12:27:12.470628Z","shell.execute_reply":"2022-03-14T12:27:16.494632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE = 0.1\n\ndf_SAMPLE = df_explo.sample(frac=SAMPLE)\n\nreducer = umap.UMAP()\nembedding = reducer.fit_transform(df_SAMPLE)\nembedding.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:38:04.794273Z","iopub.execute_input":"2022-03-14T12:38:04.794703Z","iopub.status.idle":"2022-03-14T12:38:54.124159Z","shell.execute_reply.started":"2022-03-14T12:38:04.79466Z","shell.execute_reply":"2022-03-14T12:38:54.123251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lim_q = 0.05\n\nfor c in df_SAMPLE.columns:\n    print(c)\n    plt.scatter(embedding[:, 0],embedding[:, 1], s=0.01, c = df_SAMPLE[c], vmin=df_SAMPLE[c].quantile(lim_q), vmax=df_SAMPLE[c].quantile(1-lim_q))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:51:47.528007Z","iopub.execute_input":"2022-03-14T12:51:47.52892Z","iopub.status.idle":"2022-03-14T12:52:11.99951Z","shell.execute_reply.started":"2022-03-14T12:51:47.528872Z","shell.execute_reply":"2022-03-14T12:52:11.9985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target / Feature exploration","metadata":{}},{"cell_type":"markdown","source":"For a given specie we can check how the target behave regarding the different features.","metadata":{}},{"cell_type":"code","source":"specie_id = 5045\n\ndf_merge = df_obs.merge(df_env.set_index('observation_id'), left_index=True, right_index=True, how='left')\ndf_merge['country'] = df_merge.index.astype('str').str[0]\ndf_merge['binary_target'] = df_merge.species_id == specie_id\n\nnb_q = 20\n\nfeatures = [\n    'latitude', 'longitude', 'bio_1', 'bio_2', 'bio_3', 'bio_4', 'bio_5',\n       'bio_6', 'bio_7', 'bio_8', 'bio_9', 'bio_10', 'bio_11', 'bio_12',\n       'bio_13', 'bio_14', 'bio_15', 'bio_16', 'bio_17', 'bio_18', 'bio_19',\n       'bdticm', 'bldfie', 'cecsol', 'clyppt', 'orcdrc', 'phihox', 'sltppt',\n       'sndppt'\n]\n\nfor c in features:\n    print(c)\n    quant = df_merge[c].quantile(np.arange(nb_q)/nb_q).values\n    df_merge['q'] = pd.cut(df_merge[c], quant, duplicates = 'drop')\n    avg_bin_target_by_q = df_merge.groupby('q')['binary_target'].mean()\n    avg_q_by_q = df_merge.groupby('q')[c].mean()\n    plt.scatter(avg_q_by_q.values,avg_bin_target_by_q.values)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-14T12:19:15.459175Z","iopub.execute_input":"2022-03-14T12:19:15.459603Z","iopub.status.idle":"2022-03-14T12:19:30.949475Z","shell.execute_reply.started":"2022-03-14T12:19:15.459563Z","shell.execute_reply":"2022-03-14T12:19:30.948397Z"},"trusted":true},"execution_count":null,"outputs":[]}]}