{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91196,"databundleVersionId":11432986,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Import Libraries\nimport os\nimport pandas as pd\nimport numpy as np\nimport geopandas as gpd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom shapely.geometry import Point\nimport torch\nimport plotly.express as px","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:57:56.935930Z","iopub.execute_input":"2025-03-25T02:57:56.936331Z","iopub.status.idle":"2025-03-25T02:58:00.415778Z","shell.execute_reply.started":"2025-03-25T02:57:56.936291Z","shell.execute_reply":"2025-03-25T02:58:00.414570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_meta = pd.read_csv(\"/kaggle/input/geolifeclef-2025/GLC25_PA_metadata_train.csv\")\ntest_meta = pd.read_csv(\"/kaggle/input/geolifeclef-2025/GLC25_PA_metadata_test.csv\")\nprint(\"Train shape:\", train_meta.shape, \" | Test shape:\", test_meta.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:00.417355Z","iopub.execute_input":"2025-03-25T02:58:00.418117Z","iopub.status.idle":"2025-03-25T02:58:01.851289Z","shell.execute_reply.started":"2025-03-25T02:58:00.418074Z","shell.execute_reply":"2025-03-25T02:58:01.850260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_meta.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:01.853301Z","iopub.execute_input":"2025-03-25T02:58:01.853654Z","iopub.status.idle":"2025-03-25T02:58:01.870488Z","shell.execute_reply.started":"2025-03-25T02:58:01.853626Z","shell.execute_reply":"2025-03-25T02:58:01.869373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_meta['geometry'] = train_meta.apply(lambda row: Point(row['lon'], row['lat']), axis=1)\ntest_meta['geometry'] = test_meta.apply(lambda row: Point(row['lon'], row['lat']), axis=1)\n\ntrain_gdf = gpd.GeoDataFrame(train_meta, geometry='geometry', crs='EPSG:4326')\ntest_gdf = gpd.GeoDataFrame(test_meta, geometry='geometry', crs='EPSG:4326')\n\nplt.figure(figsize=(12, 6))\nplt.scatter(train_gdf['lon'], train_gdf['lat'], s=1, label='Train', alpha=0.5, color='green')\nplt.scatter(test_gdf['lon'], test_gdf['lat'], s=1, label='Test', alpha=0.5, color='red')\nplt.legend()\nplt.title(\"Train vs Test Geospatial Distribution\")\nplt.xlabel(\"Longitude\")\nplt.ylabel(\"Latitude\")\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:01.872062Z","iopub.execute_input":"2025-03-25T02:58:01.872330Z","iopub.status.idle":"2025-03-25T02:58:44.669060Z","shell.execute_reply.started":"2025-03-25T02:58:01.872307Z","shell.execute_reply":"2025-03-25T02:58:44.667957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_train = train_meta.isnull().sum()\nmissing_train[missing_train > 0].sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:44.670187Z","iopub.execute_input":"2025-03-25T02:58:44.670493Z","iopub.status.idle":"2025-03-25T02:58:44.928769Z","shell.execute_reply.started":"2025-03-25T02:58:44.670464Z","shell.execute_reply":"2025-03-25T02:58:44.927757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"species_count = train_meta['speciesId'].value_counts()\nspecies_count_cleaned = species_count.replace([np.inf, -np.inf], np.nan).dropna()\n\nplt.figure(figsize=(14, 4))\nsns.histplot(species_count_cleaned.values, bins=100, kde=True, color=\"blue\")\nplt.title(\"Species Frequency Distribution\")\nplt.xlabel(\"Number of Observations per Species\")\nplt.ylabel(\"Species Count\")\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:44.929735Z","iopub.execute_input":"2025-03-25T02:58:44.930018Z","iopub.status.idle":"2025-03-25T02:58:45.345327Z","shell.execute_reply.started":"2025-03-25T02:58:44.929994Z","shell.execute_reply":"2025-03-25T02:58:45.344187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bioclim = pd.read_csv(\"/kaggle/input/geolifeclef-2025/EnvironmentalValues/ClimateAverage_1981-2010/GLC25-PA-test-bioclimatic.csv\")\nprint(\"Bioclim shape:\", bioclim.shape)\nbioclim.describe().T.style.background_gradient(cmap=\"YlGnBu\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:45.346403Z","iopub.execute_input":"2025-03-25T02:58:45.346687Z","iopub.status.idle":"2025-03-25T02:58:45.496162Z","shell.execute_reply.started":"2025-03-25T02:58:45.346664Z","shell.execute_reply":"2025-03-25T02:58:45.495152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the sample ID and cube path\nsample_id = 1000012\ncube_path = \"/kaggle/input/geolifeclef-2025/SateliteTimeSeries-Landsat/cubes/PA-train/GLC25-PA-train-landsat-time-series_1000012_cube.pt\"\n\n# Safely load the Landsat cube\ntry:\n    landsat_cube = torch.load(cube_path, weights_only=True)\nexcept TypeError:\n    landsat_cube = torch.load(cube_path)  # fallback for older PyTorch versions\n\nprint(\"Cube shape (bands, quarters, years):\", landsat_cube.shape)\n\n# Flatten and visualize the time series for each band\nplt.figure(figsize=(14, 5))\nbands = ['Red', 'Green', 'Blue', 'NIR', 'SWIR1', 'SWIR2']\nfor i in range(6):\n    plt.plot(landsat_cube[i].flatten().numpy(), label=bands[i])\n\nplt.title(f\"Landsat Time Series for Survey ID: {sample_id}\")\nplt.xlabel(\"Time Steps (Quarters × Years)\")\nplt.ylabel(\"Reflectance\")\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:45.498694Z","iopub.execute_input":"2025-03-25T02:58:45.499211Z","iopub.status.idle":"2025-03-25T02:58:45.828265Z","shell.execute_reply.started":"2025-03-25T02:58:45.499182Z","shell.execute_reply":"2025-03-25T02:58:45.827031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bioclim_csv = pd.read_csv(\"/kaggle/input/geolifeclef-2025/BioclimTimeSeries/values/GLC25-PA-test-bioclimatic_monthly.csv\")\n\n# Pick a survey ID\nsurvey_id = 1001615\nsample_df = bioclim_csv[bioclim_csv[\"surveyId\"] == survey_id].drop(columns=[\"surveyId\"])\n\n# Reshape to long format\nlong_df = sample_df.T.reset_index()\nlong_df.columns = [\"feature_time\", \"value\"]\n\n# Extract metadata\nlong_df[[\"feature\", \"month\", \"year\"]] = long_df[\"feature_time\"].str.extract(r'Bio-(\\w+)_([0-9]+)_([0-9]+)')\nlong_df[\"date\"] = pd.to_datetime(dict(year=long_df[\"year\"].astype(int),\n                                      month=long_df[\"month\"].astype(int),\n                                      day=1))\n\n# Plotly interactive line plot\nfig = px.line(\n    long_df,\n    x=\"date\",\n    y=\"value\",\n    color=\"feature\",\n    title=f\"🌿 Bioclimatic Features Over Time (Survey ID: {survey_id})\",\n    labels={\"date\": \"Date\", \"value\": \"Feature Value\"},\n    template=\"plotly_dark\"\n)\n\nfig.update_layout(\n    legend=dict(title=\"Feature\", orientation=\"v\", x=1.01, y=1),\n    margin=dict(l=60, r=150, t=60, b=60),\n    height=600\n)\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:45.830023Z","iopub.execute_input":"2025-03-25T02:58:45.830427Z","iopub.status.idle":"2025-03-25T02:58:47.760307Z","shell.execute_reply.started":"2025-03-25T02:58:45.830386Z","shell.execute_reply":"2025-03-25T02:58:47.759221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load metadatasets\nmeta = pd.read_csv(\"/kaggle/input/geolifeclef-2025/GLC25_PA_metadata_train.csv\")\nelev = pd.read_csv(\"/kaggle/input/geolifeclef-2025/EnvironmentalValues/Elevation/GLC25-PA-train-elevation.csv\")\nfoot = pd.read_csv(\"/kaggle/input/geolifeclef-2025/EnvironmentalValues/HumanFootprint/GLC25-PA-train-human_footprint.csv\")\n\n# Preprocess metadata: keep unique surveys\nmeta = meta.drop_duplicates(subset=\"surveyId\")\n\n# Rename elevation column if necessary\nif \"value\" in elev.columns:\n    elev = elev.rename(columns={\"value\": \"Elevation\"})\n\n# Choose the footprint feature to visualize\nselected_footprint = \"HumanFootprint-building-residential\"  # ← swap to any other as needed\n\n# Reduce footprint to just selected feature\nfoot = foot[[\"surveyId\", selected_footprint]].rename(columns={selected_footprint: \"FootprintFeature\"})\n\n# Merge everything together\ndf = meta.merge(elev, on=\"surveyId\", how=\"left\").merge(foot, on=\"surveyId\", how=\"left\")\n\n# Drop missing values\ndf = df.dropna(subset=[\"Elevation\", \"FootprintFeature\", \"lat\"])\n\n# Plotly 3D scatter\nfig = px.scatter_3d(\n    df,\n    x=\"FootprintFeature\",\n    y=\"Elevation\",\n    z=\"lat\",\n    color=\"region\",\n    hover_data=[\"country\", \"year\", \"lon\", \"lat\", \"speciesId\"],\n    title=f\"🌍 3D Landscape View: {selected_footprint.split('-')[-1].capitalize()} vs Elevation vs Latitude\",\n    opacity=0.75,\n    height=700\n)\n\nfig.update_layout(scene=dict(\n    xaxis_title=selected_footprint.split('-')[-1].capitalize() + \" (%)\",\n    yaxis_title=\"Elevation (m)\",\n    zaxis_title=\"Latitude\"\n))\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:47.761432Z","iopub.execute_input":"2025-03-25T02:58:47.761776Z","iopub.status.idle":"2025-03-25T02:58:50.387620Z","shell.execute_reply.started":"2025-03-25T02:58:47.761740Z","shell.execute_reply":"2025-03-25T02:58:50.385804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean column names\nelev = elev.rename(columns={'value': 'Elevation'}) if 'value' in elev.columns else elev\nfoot = foot.rename(columns={'value': 'HumanFootprint'}) if 'value' in foot.columns else foot\n\n# Merge on unique surveyId level\nmeta_unique = meta.drop_duplicates(\"surveyId\")\ndf = meta_unique.merge(elev, on=\"surveyId\", how=\"left\").merge(foot, on=\"surveyId\", how=\"left\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:50.388824Z","iopub.execute_input":"2025-03-25T02:58:50.389159Z","iopub.status.idle":"2025-03-25T02:58:50.412350Z","shell.execute_reply.started":"2025-03-25T02:58:50.389131Z","shell.execute_reply":"2025-03-25T02:58:50.411357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"heatmap = px.density_mapbox(\n    df, lat=\"lat\", lon=\"lon\", radius=5,\n    center={\"lat\": 47, \"lon\": 10}, zoom=3,\n    mapbox_style=\"carto-positron\",\n    title=\"🌍 Survey Site Density Heatmap\"\n)\nheatmap.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:50.413273Z","iopub.execute_input":"2025-03-25T02:58:50.413533Z","iopub.status.idle":"2025-03-25T02:58:50.510378Z","shell.execute_reply.started":"2025-03-25T02:58:50.413511Z","shell.execute_reply":"2025-03-25T02:58:50.509125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"species_rich = meta.groupby(\"surveyId\")[\"speciesId\"].nunique().reset_index()\nspecies_rich = species_rich.merge(df[[\"surveyId\", \"lat\", \"lon\"]], on=\"surveyId\")\n\nbubble_map = px.scatter_mapbox(\n    species_rich, lat=\"lat\", lon=\"lon\", size=\"speciesId\",\n    color=\"speciesId\", zoom=3, size_max=10,\n    mapbox_style=\"open-street-map\",\n    title=\"🧪 Species Richness per Survey Location\",\n    labels={\"speciesId\": \"Species Count\"}\n)\nbubble_map.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:50.511508Z","iopub.execute_input":"2025-03-25T02:58:50.511860Z","iopub.status.idle":"2025-03-25T02:58:50.631742Z","shell.execute_reply.started":"2025-03-25T02:58:50.511825Z","shell.execute_reply":"2025-03-25T02:58:50.630431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lat_div = meta.groupby(\"lat\")[\"speciesId\"].nunique().reset_index()\nlat_div.columns = [\"Latitude\", \"UniqueSpecies\"]\n\nlat_plot = px.line(\n    lat_div, x=\"Latitude\", y=\"UniqueSpecies\",\n    title=\"📊 Species Diversity by Latitude\",\n    labels={\"UniqueSpecies\": \"Unique Species Count\"}\n)\nlat_plot.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:50.632925Z","iopub.execute_input":"2025-03-25T02:58:50.633327Z","iopub.status.idle":"2025-03-25T02:58:50.740052Z","shell.execute_reply.started":"2025-03-25T02:58:50.633289Z","shell.execute_reply":"2025-03-25T02:58:50.738944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"🔢 Total unique surveyIds:\", meta[\"surveyId\"].nunique())\nprint(\"🧬 Total unique species:\", meta[\"speciesId\"].nunique())\nprint(\"🌍 Geographic Range:\")\nprint(f\"    Latitude: {meta['lat'].min():.2f} → {meta['lat'].max():.2f}\")\nprint(f\"    Longitude: {meta['lon'].min():.2f} → {meta['lon'].max():.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:58:50.741265Z","iopub.execute_input":"2025-03-25T02:58:50.741674Z","iopub.status.idle":"2025-03-25T02:58:50.759144Z","shell.execute_reply.started":"2025-03-25T02:58:50.741633Z","shell.execute_reply":"2025-03-25T02:58:50.757970Z"}},"outputs":[],"execution_count":null}]}