{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91196,"databundleVersionId":11432986,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Published on March 20, 2024. By Marília Prata, mpwolke","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\n\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport plotly.graph_objects as go\nimport plotly.offline as py\nimport plotly.express as px\nfrom plotly.offline import iplot\n\n#Ignore warnings\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-20T01:50:17.764169Z","iopub.execute_input":"2025-03-20T01:50:17.764481Z","execution_failed":"2025-03-20T01:52:11.708Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"![](https://storage.googleapis.com/kaggle-competitions/kaggle/91196/logos/header.png?GoogleAccessId=web-data@kaggle-161607.iam.gserviceaccount.com&Expires=1742492977&Signature=ceu0YryZP8cbkMWDFw%2BJ6ZHfIZ47QeMvwGJNh16SOalQN0JiBevUzNSOMx7fr5c9NMTYWwEIyyq7qLcsvxl5%2BJjEZQKsvbuq7eT2EcO9mBqsczt8jFxDSYKME%2BQUX5l6EofEaaYJYnUjv%2F4aBxKxo6pAg8c78kqkFg5agIr3JFriOXgYzVyNi%2BgHmo9axnoAKsQK0TIFGqd34BfWt6H0YPfHML5oKQnw677mwv%2BYhS25K9YgohF89U3A0vCH16UP4%2BZG%2BjPKTl0ZSZ7P1MCkrdYOtzTLhhpF9NJkREO97wGm0g9wGz7eQji38aSDeucGmDdVA0eduFzleEiUiwyE%2FQ%3D%3D)","metadata":{}},{"cell_type":"markdown","source":"## Competition Citation\n\n@misc{geolifeclef-2025,\n\n    author = {Alexis Joly and César Leblanc and DZombie and Maximilien Servajean and picekl and tlarcher},\n    \n    title = {GeoLifeCLEF25 @ CVPR & LifeCLEF},\n    \n    year = {2025},\n    howpublished = {\\url{https://kaggle.com/competitions/geolifeclef-2025}},","metadata":{}},{"cell_type":"markdown","source":"### Overview\n\n\"This challenge aims to predict plant species in a given location and time using various possible predictors: satellite images and time series, climatic time series, and other rasterized environmental data: land cover, human footprint, bioclimatic, and soil variables.\"\n\nhttps://www.kaggle.com/competitions/geolifeclef-2025/overview","metadata":{}},{"cell_type":"markdown","source":"### Import Libraries","metadata":{}},{"cell_type":"code","source":"#By Go Byeonggeon https://www.kaggle.com/code/gobyeonggeon/preprocess-visualize-spatial-data-eda-xgb/notebook\n\nfrom glob import glob\nimport os\n# from netCDF4 import Dataset\nimport pandas as pd\nimport geopandas as gpd #\nfrom shapely.geometry import Polygon, LineString, Point\n#import rasterio #No module named 'rasterio'\n#from rasterio.plot import show\n# from rasterio.transform import from_origin\n#from rasterstats import zonal_stats\nimport matplotlib.pyplot as plt\nimport cv2\nimport numpy as np\nimport geopandas as gpd\nimport tqdm\n\nroot_path = '/kaggle/input'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T02:08:11.131542Z","iopub.execute_input":"2025-03-20T02:08:11.131924Z","iopub.status.idle":"2025-03-20T02:08:11.548176Z","shell.execute_reply.started":"2025-03-20T02:08:11.131892Z","shell.execute_reply":"2025-03-20T02:08:11.546691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Go Byeonggeon https://www.kaggle.com/code/gobyeonggeon/preprocess-visualize-spatial-data-eda-xgb/notebook\n\ntrain_meta = pd.read_csv(root_path + \"/geolifeclef-2025/GLC25_PA_metadata_train.csv\")\ntrain_meta.head()\n\nsub_train_meta = train_meta.drop_duplicates('surveyId').sample(n=1000, random_state=42)\nsub_train_meta.index = range(len(sub_train_meta))\n\n# make Point vector \npoint_list = []\nfor i in tqdm.tqdm(range(len(sub_train_meta))):\n    x,y = sub_train_meta.loc[i, ['lon', 'lat']]\n    poind_i = Point(x,y)\n    point_list.append(poind_i)\n\nsub_train_meta.loc[:,'geometry'] = point_list\n\n# Read meta data\ntest_meta = pd.read_csv(root_path + \"/geolifeclef-2025/GLC25_PA_metadata_test.csv\")\ntest_meta.head()\n\nsub_test_meta = test_meta.drop_duplicates('surveyId').sample(n=1000, random_state=42)\nsub_test_meta.index = range(len(sub_test_meta))\n\n# make Point vector \npoint_list = []\nfor i in tqdm.tqdm(range(len(sub_test_meta))):\n    x,y = sub_test_meta.loc[i, ['lon', 'lat']]\n    poind_i = Point(x,y)\n    point_list.append(poind_i)\n\nsub_test_meta.loc[:,'geometry'] = point_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T02:12:31.694723Z","iopub.execute_input":"2025-03-20T02:12:31.695140Z","iopub.status.idle":"2025-03-20T02:12:34.057354Z","shell.execute_reply.started":"2025-03-20T02:12:31.695107Z","shell.execute_reply":"2025-03-20T02:12:34.055865Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Install Folium Matplotlib Mapclassify","metadata":{}},{"cell_type":"code","source":"!pip install folium matplotlib mapclassify","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T02:14:32.656921Z","iopub.execute_input":"2025-03-20T02:14:32.657294Z","iopub.status.idle":"2025-03-20T02:14:38.749723Z","shell.execute_reply.started":"2025-03-20T02:14:32.657268Z","shell.execute_reply":"2025-03-20T02:14:38.748193Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Folium Map with Geometry","metadata":{}},{"cell_type":"code","source":"#By Go Byeonggeon https://www.kaggle.com/code/gobyeonggeon/preprocess-visualize-spatial-data-eda-xgb/notebook\n\ntrain_sub_meta_gdf = gpd.GeoDataFrame(sub_train_meta, geometry = 'geometry')\n#train_sub_meta_gdf.crs = {'init':'epsg:4326'}#'+init=<authority>:<code>' syntax is deprecated.\ntrain_sub_meta_gdf.crs = ('EPSG:4326')\ntrain_sub_meta_gdf.head()\n\ntest_sub_meta_gdf = gpd.GeoDataFrame(sub_test_meta, geometry = 'geometry')\n#test_sub_meta_gdf.crs = {'init':'epsg:4326'} #'+init=<authority>:<code>' syntax is deprecated.\ntest_sub_meta_gdf.crs = ('EPSG:4326') \ntest_sub_meta_gdf.head()\n\n# visualize each monitoring site\nm = train_sub_meta_gdf.drop_duplicates(['lon', 'lat']).explore(color = 'green')\ntest_sub_meta_gdf.drop_duplicates(['lon', 'lat']).explore(m=m, color = 'red')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T02:20:54.454393Z","iopub.execute_input":"2025-03-20T02:20:54.454772Z","iopub.status.idle":"2025-03-20T02:20:54.799222Z","shell.execute_reply.started":"2025-03-20T02:20:54.454743Z","shell.execute_reply":"2025-03-20T02:20:54.797700Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### It was supposed to merge the files (train/test). I did Not.\n\nI simply found (apparently) the columns with missing values to plot the Maps below.  A simply isnull().sum() only in train or info() would return that information. However, it was expected to use MERGED data.","metadata":{}},{"cell_type":"code","source":"!pip install rasterio","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T02:22:59.034423Z","iopub.execute_input":"2025-03-20T02:22:59.034815Z","iopub.status.idle":"2025-03-20T02:23:05.183725Z","shell.execute_reply.started":"2025-03-20T02:22:59.034764Z","shell.execute_reply":"2025-03-20T02:23:05.182309Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing columns should be from the Merged files that I didn't merge :(","metadata":{}},{"cell_type":"code","source":"#By Go Byeonggeon https://www.kaggle.com/code/gobyeonggeon/preprocess-visualize-spatial-data-eda-xgb/notebook\n\n# check missing values\nmissing_value_col_list = []\nrows_with_missing_values = {}\nfor column in train_meta.columns:\n    missing_rows = train_meta.index[train_meta[column].isnull()].tolist()\n    if len(missing_rows) > 0:\n        rows_with_missing_values[column] = len(missing_rows)\n        missing_value_col_list.append(column)\n\nprint(rows_with_missing_values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T02:28:55.104895Z","iopub.execute_input":"2025-03-20T02:28:55.105280Z","iopub.status.idle":"2025-03-20T02:28:55.277834Z","shell.execute_reply.started":"2025-03-20T02:28:55.105254Z","shell.execute_reply":"2025-03-20T02:28:55.276597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Go Byeonggeon https://www.kaggle.com/code/gobyeonggeon/preprocess-visualize-spatial-data-eda-xgb/notebook\n\ntarget_colnames = 'geoUncertaintyInM'\nsub_df = pd.merge(train_meta.loc[:, ['surveyId', target_colnames]], train_meta.loc[:, ['lon', 'lat', 'surveyId']].drop_duplicates(\"surveyId\"), on='surveyId')\nna_ind = sub_df.loc[:,target_colnames].isna().values\n\nsub_df_train = sub_df.loc[~na_ind,:]\nsub_df_target = sub_df.loc[na_ind,:]\nx,y,z = sub_df_train['lon'].values, sub_df_train['lat'].values, sub_df_train[target_colnames].values\nx_target,y_target,z_target = sub_df_target['lon'].values, sub_df_target['lat'].values, np.zeros(sub_df_target[target_colnames].shape)\n\nplt.figure(figsize=(12, 6))\nplt.scatter(x, y, marker='o', color='green', label='Non Missing Data')\nplt.scatter(x_target, y_target, marker='o', color='red', label='Missing Data', s = 5)\n\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T02:34:31.628065Z","iopub.execute_input":"2025-03-20T02:34:31.628438Z","iopub.status.idle":"2025-03-20T02:34:42.891454Z","shell.execute_reply.started":"2025-03-20T02:34:31.628412Z","shell.execute_reply":"2025-03-20T02:34:42.890367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Go Byeonggeon https://www.kaggle.com/code/gobyeonggeon/preprocess-visualize-spatial-data-eda-xgb/notebook\n\ntarget_colnames = 'areaInM2'\nsub_df = pd.merge(train_meta.loc[:, ['surveyId', target_colnames]], train_meta.loc[:, ['lon', 'lat', 'surveyId']].drop_duplicates(\"surveyId\"), on='surveyId')\nna_ind = sub_df.loc[:,target_colnames].isna().values\n\nsub_df_train = sub_df.loc[~na_ind,:]\nsub_df_target = sub_df.loc[na_ind,:]\nx,y,z = sub_df_train['lon'].values, sub_df_train['lat'].values, sub_df_train[target_colnames].values\nx_target,y_target,z_target = sub_df_target['lon'].values, sub_df_target['lat'].values, np.zeros(sub_df_target[target_colnames].shape)\n\nplt.figure(figsize=(12, 6))\nplt.scatter(x, y, marker='o', color='green', label='Non Missing Data')\nplt.scatter(x_target, y_target, marker='o', color='red', label='Missing Data', s = 5)\n\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T02:35:56.727164Z","iopub.execute_input":"2025-03-20T02:35:56.727511Z","iopub.status.idle":"2025-03-20T02:36:07.773954Z","shell.execute_reply.started":"2025-03-20T02:35:56.727484Z","shell.execute_reply":"2025-03-20T02:36:07.772647Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#Acknowledgements:\n\nGo Byeonggeon https://www.kaggle.com/code/gobyeonggeon/preprocess-visualize-spatial-data-eda-xgb/notebook\n\nmpwolke https://www.kaggle.com/code/mpwolke/glc2024-climaterasters-nir-pt-files/notebook","metadata":{}}]}