{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91196,"databundleVersionId":11432986,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q folium matplotlib mapclassify","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:15:47.949454Z","iopub.execute_input":"2025-03-25T02:15:47.949859Z","iopub.status.idle":"2025-03-25T02:15:51.372076Z","shell.execute_reply.started":"2025-03-25T02:15:47.949821Z","shell.execute_reply":"2025-03-25T02:15:51.370853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 📚 Core Libraries\nimport os, random, gc, math\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\n# 📊 Visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport folium\nimport geopandas as gpd\nfrom shapely.geometry import Point\n\n# 🔥 Torch Setup\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.models as models\nimport torchvision.transforms as T\n\n# ⚙️ Configs\nSEED = 42\nBATCH_SIZE = 64\nNUM_CLASSES = 11255\nEPOCHS = 30\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nDATA_PATH = \"/kaggle/input/geolifeclef-2025\"\ntorch.manual_seed(SEED)\nnp.random.seed(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:15:51.373621Z","iopub.execute_input":"2025-03-25T02:15:51.373997Z","iopub.status.idle":"2025-03-25T02:15:55.454924Z","shell.execute_reply.started":"2025-03-25T02:15:51.373958Z","shell.execute_reply":"2025-03-25T02:15:55.454199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Load metadata\ntrain_meta = pd.read_csv(\"/kaggle/input/geolifeclef-2025/GLC25_PA_metadata_train.csv\")\ntest_meta  = pd.read_csv(\"/kaggle/input/geolifeclef-2025/GLC25_PA_metadata_test.csv\")\n\n# ✅ Convert to GeoDataFrames\ntrain_meta = train_meta.drop_duplicates(\"surveyId\").copy()\ntrain_meta[\"geometry\"] = train_meta.apply(lambda x: Point(x[\"lon\"], x[\"lat\"]), axis=1)\ntrain_gdf = gpd.GeoDataFrame(train_meta, geometry=\"geometry\", crs=\"EPSG:4326\")\n\ntest_meta = test_meta.drop_duplicates(\"surveyId\").copy()\ntest_meta[\"geometry\"] = test_meta.apply(lambda x: Point(x[\"lon\"], x[\"lat\"]), axis=1)\ntest_gdf = gpd.GeoDataFrame(test_meta, geometry=\"geometry\", crs=\"EPSG:4326\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:15:55.456342Z","iopub.execute_input":"2025-03-25T02:15:55.456840Z","iopub.status.idle":"2025-03-25T02:15:58.736466Z","shell.execute_reply.started":"2025-03-25T02:15:55.456814Z","shell.execute_reply":"2025-03-25T02:15:58.735774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import geopandas as gpd\nfrom shapely.geometry import Point\n\ntrain_meta = train_meta.copy()\ntrain_meta[\"geometry\"] = train_meta.apply(lambda x: Point(x[\"lon\"], x[\"lat\"]), axis=1)\ntrain_gdf = gpd.GeoDataFrame(train_meta, geometry=\"geometry\", crs=\"EPSG:4326\")\n\n# Plot interactive map\nm = train_gdf.explore(\n    color=\"green\",\n    tiles=\"Stamen Terrain\",\n    attr='Map tiles by Stamen Design, under CC BY 3.0. Data by OpenStreetMap, under ODbL.'\n)\ntest_gdf.explore(m=m, color=\"red\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:15:58.737571Z","iopub.execute_input":"2025-03-25T02:15:58.737843Z","iopub.status.idle":"2025-03-25T02:16:15.285500Z","shell.execute_reply.started":"2025-03-25T02:15:58.737819Z","shell.execute_reply":"2025-03-25T02:16:15.283923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class GeoLifeDataset(Dataset):\n    def __init__(self, meta_df, cube_path, subset, transform=None):\n        self.meta = meta_df.dropna(subset=['speciesId']).drop_duplicates(\"surveyId\")\n        self.label_dict = self.meta.groupby(\"surveyId\")[\"speciesId\"].apply(list).to_dict()\n        self.cube_path = cube_path\n        self.transform = transform\n        self.subset = subset\n\n    def __len__(self): return len(self.meta)\n\n    def __getitem__(self, idx):\n        survey_id = self.meta.iloc[idx].surveyId\n        cube = torch.nan_to_num(torch.load(f\"{self.cube_path}/GLC25-PA-{self.subset}-landsat_time_series_{survey_id}_cube.pt\", weights_only=True))\n        cube = cube.permute(1, 2, 0).numpy()  # HWC\n        label = torch.zeros(NUM_CLASSES)\n        for sid in self.label_dict.get(survey_id, []):\n            label[sid] = 1\n        if self.transform: cube = self.transform(cube)\n        return cube, label, survey_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:16:15.286530Z","iopub.execute_input":"2025-03-25T02:16:15.286761Z","iopub.status.idle":"2025-03-25T02:16:15.295648Z","shell.execute_reply.started":"2025-03-25T02:16:15.286740Z","shell.execute_reply":"2025-03-25T02:16:15.294668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ModifiedResNet18(nn.Module):\n    def __init__(self, num_classes):\n        super().__init__()\n        self.resnet = models.resnet18(weights=None)\n        self.resnet.conv1 = nn.Conv2d(6, 64, kernel_size=3, stride=1, padding=1, bias=False)\n        self.resnet.maxpool = nn.Identity()\n        self.ln = nn.LayerNorm(1000)\n        self.fc = nn.Sequential(\n            nn.Linear(1000, 2048),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(2048, num_classes)\n        )\n\n    def forward(self, x):\n        x = x.permute(0, 3, 1, 2).float()  # NHWC -> NCHW\n        x = self.resnet(x)\n        x = self.ln(x)\n        return self.fc(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:16:15.296717Z","iopub.execute_input":"2025-03-25T02:16:15.297071Z","iopub.status.idle":"2025-03-25T02:16:15.315326Z","shell.execute_reply.started":"2025-03-25T02:16:15.297022Z","shell.execute_reply":"2025-03-25T02:16:15.314269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_optimizer(model):\n    return torch.optim.AdamW(model.parameters(), lr=2e-4, weight_decay=1e-4)\n\ndef get_scheduler(optimizer):\n    return torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=EPOCHS)\n\ndef get_loss(targets, logits):\n    pos_weight = targets * 1.0\n    criterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight)\n    return criterion(logits, targets)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:16:15.316495Z","iopub.execute_input":"2025-03-25T02:16:15.316889Z","iopub.status.idle":"2025-03-25T02:16:15.324846Z","shell.execute_reply.started":"2025-03-25T02:16:15.316849Z","shell.execute_reply":"2025-03-25T02:16:15.323884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.cuda.amp import autocast, GradScaler\n\ndef train_one_epoch(model, loader, optimizer, scaler):\n    model.train()\n    total_loss = 0\n    for data, target, _ in tqdm(loader):\n        data, target = data.to(DEVICE), target.to(DEVICE)\n        optimizer.zero_grad()\n        with autocast():\n            output = model(data)\n            loss = get_loss(target, output)\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n        total_loss += loss.item()\n    return total_loss / len(loader)\n\ndef train_model(model, train_loader):\n    optimizer = get_optimizer(model)\n    scheduler = get_scheduler(optimizer)\n    scaler = GradScaler()\n    for epoch in range(EPOCHS):\n        loss = train_one_epoch(model, train_loader, optimizer, scaler)\n        print(f\"Epoch {epoch+1}/{EPOCHS} - Loss: {loss:.4f}\")\n        scheduler.step()\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:16:15.327997Z","iopub.execute_input":"2025-03-25T02:16:15.328418Z","iopub.status.idle":"2025-03-25T02:16:15.339456Z","shell.execute_reply.started":"2025-03-25T02:16:15.328377Z","shell.execute_reply":"2025-03-25T02:16:15.338174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class GeoLifeDataset(Dataset):\n    def __init__(self, meta_df, cube_path, subset, transform=None):\n        self.meta = meta_df.dropna(subset=['speciesId']).drop_duplicates(\"surveyId\")\n        self.label_dict = self.meta.groupby(\"surveyId\")[\"speciesId\"].apply(list).to_dict()\n        self.cube_path = cube_path\n        self.transform = transform\n        self.subset = subset\n\n    def __len__(self):\n        return len(self.meta)\n\n    def __getitem__(self, idx):\n        survey_id = self.meta.iloc[idx].surveyId\n        cube = torch.nan_to_num(torch.load(f\"{self.cube_path}/GLC25-PA-{self.subset}-landsat_time_series_{survey_id}_cube.pt\", weights_only=True))\n        cube = cube.permute(1, 2, 0).numpy()  # HWC\n        label = torch.zeros(11255)\n        for sid in self.label_dict.get(survey_id, []):\n            label[sid] = 1\n        if self.transform:\n            cube = self.transform(cube)\n        return cube, label, survey_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T02:16:15.340514Z","iopub.execute_input":"2025-03-25T02:16:15.340896Z","iopub.status.idle":"2025-03-25T02:16:15.352234Z","shell.execute_reply.started":"2025-03-25T02:16:15.340857Z","shell.execute_reply":"2025-03-25T02:16:15.351135Z"}},"outputs":[],"execution_count":null}]}