{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":121548,"databundleVersionId":14579951,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:28:34.376590Z","iopub.execute_input":"2025-11-22T09:28:34.376871Z","iopub.status.idle":"2025-11-22T09:28:50.072748Z","shell.execute_reply.started":"2025-11-22T09:28:34.376839Z","shell.execute_reply":"2025-11-22T09:28:50.071448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\nfrom PIL import Image\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport time\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:28:50.074355Z","iopub.execute_input":"2025-11-22T09:28:50.075341Z","iopub.status.idle":"2025-11-22T09:28:57.325602Z","shell.execute_reply.started":"2025-11-22T09:28:50.075305Z","shell.execute_reply":"2025-11-22T09:28:57.324980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nBASE_DIR = '/kaggle/input/helipad-detection-challenge-sup-com/helipad_hackathon/'\nTRAIN_CSV = os.path.join(BASE_DIR, 'train.csv')\nIMAGE_DIR = os.path.join(BASE_DIR, 'images')\n\n# Hyperparameters\nBATCH_SIZE = 32\nLEARNING_RATE = 1e-4\nEPOCHS = 12\nIMG_SIZE = 224\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nprint(f\"Training on: {DEVICE}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:28:57.326393Z","iopub.execute_input":"2025-11-22T09:28:57.326773Z","iopub.status.idle":"2025-11-22T09:28:57.381742Z","shell.execute_reply.started":"2025-11-22T09:28:57.326740Z","shell.execute_reply":"2025-11-22T09:28:57.380995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass HelipadDataset(Dataset):\n    def __init__(self, df, root_dir, transform=None):\n        self.df = df\n        self.root_dir = root_dir\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        img_name = self.df.iloc[idx, 0] \n        label = self.df.iloc[idx, 1]\n        img_path = os.path.join(self.root_dir, img_name)\n        \n        try:\n            image = Image.open(img_path).convert(\"RGB\")\n        except:\n            image = Image.new('RGB', (IMG_SIZE, IMG_SIZE), (0, 0, 0))\n        \n        if self.transform:\n            image = self.transform(image)\n            \n        return image, torch.tensor(label, dtype=torch.float32)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:28:57.383582Z","iopub.execute_input":"2025-11-22T09:28:57.383798Z","iopub.status.idle":"2025-11-22T09:28:57.399994Z","shell.execute_reply.started":"2025-11-22T09:28:57.383780Z","shell.execute_reply":"2025-11-22T09:28:57.399129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. PREPARE DATA & AUGMENTATION ---\nfull_df = pd.read_csv(TRAIN_CSV)\n\ntrain_df, val_df = train_test_split(\n    full_df, \n    test_size=0.15, \n    stratify=full_df['label'], \n    random_state=42\n)\n\nprint(f\"Training Images:   {len(train_df)}\")\nprint(f\"Validation Images: {len(val_df)}\")\n\ntrain_transforms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.RandomHorizontalFlip(0.5),\n    transforms.RandomVerticalFlip(0.5),\n    transforms.RandomRotation(15),\n    transforms.ColorJitter(brightness=0.1, contrast=0.1),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n])\n\nval_transforms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:28:57.400913Z","iopub.execute_input":"2025-11-22T09:28:57.401295Z","iopub.status.idle":"2025-11-22T09:28:57.458220Z","shell.execute_reply.started":"2025-11-22T09:28:57.401268Z","shell.execute_reply":"2025-11-22T09:28:57.457563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_loader = DataLoader(\n    HelipadDataset(train_df, IMAGE_DIR, train_transforms), \n    batch_size=BATCH_SIZE, shuffle=True, num_workers=2\n)\n\nval_loader = DataLoader(\n    HelipadDataset(val_df, IMAGE_DIR, val_transforms), \n    batch_size=BATCH_SIZE, shuffle=False, num_workers=2\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:28:57.458909Z","iopub.execute_input":"2025-11-22T09:28:57.459134Z","iopub.status.idle":"2025-11-22T09:28:57.464008Z","shell.execute_reply.started":"2025-11-22T09:28:57.459119Z","shell.execute_reply":"2025-11-22T09:28:57.463208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 3. LOAD RESNET50 MODEL ---\nfrom torchvision.models import ResNet50_Weights\n\nmodel = models.resnet50(weights=ResNet50_Weights.IMAGENET1K_V1)\n\n\nnum_features = model.fc.in_features \nmodel.fc = nn.Sequential(\n    nn.Linear(num_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.4),\n    nn.Linear(512, 1)\n)\n\nmodel = model.to(DEVICE)\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.AdamW(model.parameters(), lr=LEARNING_RATE, weight_decay=1e-4)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:28:57.464702Z","iopub.execute_input":"2025-11-22T09:28:57.464994Z","iopub.status.idle":"2025-11-22T09:28:58.878513Z","shell.execute_reply.started":"2025-11-22T09:28:57.464970Z","shell.execute_reply":"2025-11-22T09:28:58.877918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 4. TRAINING LOOP ---\nbest_val_acc = 0.0\nhistory = {'train_loss': [], 'val_acc': []}\n\nprint(\"\\n--- Starting Training ---\")\nstart_time = time.time()\n\nfor epoch in range(EPOCHS):\n    model.train()\n    running_loss = 0.0\n    \n    for images, labels in train_loader:\n        images, labels = images.to(DEVICE), labels.to(DEVICE)\n        labels = labels.unsqueeze(1)\n        \n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n    \n    epoch_loss = running_loss / len(train_loader)\n    history['train_loss'].append(epoch_loss)\n    \n    # VALIDATION\n    model.eval()\n    correct = 0\n    total = 0\n    \n    with torch.no_grad():\n        for images, labels in val_loader:\n            images, labels = images.to(DEVICE), labels.to(DEVICE)\n            labels = labels.unsqueeze(1)\n            \n            outputs = model(images)\n            preds = torch.sigmoid(outputs)\n            predicted_labels = (preds > 0.5).float()\n            \n            correct += (predicted_labels == labels).sum().item()\n            total += labels.size(0)\n    \n    epoch_acc = correct / total\n    history['val_acc'].append(epoch_acc)\n    \n    print(f\"Epoch {epoch+1}/{EPOCHS} | Loss: {epoch_loss:.4f} | Val Acc: {epoch_acc:.4f}\")\n    \n    if epoch_acc > best_val_acc:\n        best_val_acc = epoch_acc\n        torch.save(model.state_dict(), 'helipad_resnet50.pth')\n        print(\"  >>> Model Saved (New Best)\")\n\ntotal_time = (time.time() - start_time) / 60\nprint(f\"\\nTraining Finished in {total_time:.1f} minutes.\")\nprint(f\"Best Validation Accuracy: {best_val_acc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:28:58.879274Z","iopub.execute_input":"2025-11-22T09:28:58.879518Z","iopub.status.idle":"2025-11-22T09:45:19.047228Z","shell.execute_reply.started":"2025-11-22T09:28:58.879501Z","shell.execute_reply":"2025-11-22T09:45:19.046312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 5. PLOT RESULTS ---\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(history['train_loss'], label='Train Loss', color='red')\nplt.title('Training Loss')\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nplt.plot(history['val_acc'], label='Validation Accuracy', color='blue')\nplt.title('Validation Accuracy')\nplt.grid(True)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:45:34.047009Z","iopub.execute_input":"2025-11-22T09:45:34.047831Z","iopub.status.idle":"2025-11-22T09:45:34.379760Z","shell.execute_reply.started":"2025-11-22T09:45:34.047801Z","shell.execute_reply":"2025-11-22T09:45:34.378974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\nfrom PIL import Image\nfrom tqdm import tqdm # Progress bar\n\n# --- CONFIGURATION ---\nBASE_DIR = '/kaggle/input/helipad-detection-challenge-sup-com/helipad_hackathon/'\nTEST_CSV_PATH = os.path.join(BASE_DIR, 'test.csv') \nIMAGE_DIR = os.path.join(BASE_DIR, 'images')\nMODEL_PATH = 'helipad_resnet50.pth' # The file you just trained\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nIMG_SIZE = 224\n\n# --- 1. TEST DATASET CLASS ---\n# This is different from training because we don't have labels\nclass TestDataset(Dataset):\n    def __init__(self, df, root_dir, transform=None):\n        self.df = df\n        self.root_dir = root_dir\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        # Get the Image ID (assuming column 0 is the ID)\n        img_name = str(self.df.iloc[idx, 0])\n        \n        # Handle file paths safely\n        img_path = os.path.join(self.root_dir, img_name)\n        \n        # If extension is missing in CSV, try adding it\n        if not os.path.exists(img_path):\n            if os.path.exists(img_path + '.jpg'): img_path += '.jpg'\n            elif os.path.exists(img_path + '.png'): img_path += '.png'\n        \n        try:\n            image = Image.open(img_path).convert(\"RGB\")\n        except:\n            # If image is missing, return a black image (rare safety check)\n            image = Image.new('RGB', (IMG_SIZE, IMG_SIZE), (0, 0, 0))\n            \n        if self.transform:\n            image = self.transform(image)\n            \n        return image, img_name\n\n# --- 2. SETUP ---\n# Load the list of test images\n# If test.csv doesn't exist, check for sample_submission.csv\nif not os.path.exists(TEST_CSV_PATH):\n    TEST_CSV_PATH = os.path.join(BASE_DIR, 'sample_submission.csv')\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\nprint(f\"Loaded Test CSV: {len(test_df)} images to predict.\")\n\n# Define Transforms (Must be same resizing/normalization as training)\ntest_transforms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n])\n\n# Create Loader\ntest_dataset = TestDataset(test_df, IMAGE_DIR, transform=test_transforms)\ntest_loader = DataLoader(test_dataset, batch_size=32, shuffle=False, num_workers=2)\n\n# --- 3. LOAD MODEL ARCHITECTURE ---\n# We must define the model EXACTLY as we did in training\nmodel = models.resnet50(pretrained=False) # Pretrained=False because we load our own weights\n\n# Re-create the custom head\nnum_features = model.fc.in_features\nmodel.fc = nn.Sequential(\n    nn.Linear(num_features, 512),\n    nn.ReLU(),\n    nn.Dropout(0.4),\n    nn.Linear(512, 1)\n)\n\n# Load the trained weights\nif os.path.exists(MODEL_PATH):\n    model.load_state_dict(torch.load(MODEL_PATH, map_location=DEVICE))\n    print(\"✅ Model weights loaded successfully!\")\nelse:\n    print(f\"❌ ERROR: Could not find {MODEL_PATH}. Did you run the training cell?\")\n\nmodel = model.to(DEVICE)\nmodel.eval() # Important: Turns off Dropout and BatchNorm updates\n\n# --- 4. PREDICTION LOOP ---\npredictions = []\nimage_ids = []\n\nprint(\"Starting Prediction...\")\n\nwith torch.no_grad(): # Disable gradient calculation to save memory\n    for images, names in tqdm(test_loader):\n        images = images.to(DEVICE)\n        \n        # Forward pass\n        outputs = model(images)\n        \n        # Convert logits to probabilities (0 to 1)\n        probs = torch.sigmoid(outputs)\n        \n        # Move to CPU and store\n        predictions.extend(probs.cpu().numpy().flatten())\n        image_ids.extend(names)\n\n# --- 5. CREATE SUBMISSION CSV ---\n# Apply Threshold (0.5 is standard)\nbinary_predictions = [1 if p > 0.5 else 0 for p in predictions]\n\n# Create DataFrame\nsubmission_df = pd.DataFrame({\n    'id': image_ids,      # Ensure this column name matches Kaggle's sample_submission\n    'label': binary_predictions\n})\n\n# Save\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"\\n✅ 'submission.csv' created successfully!\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:52:03.317775Z","iopub.execute_input":"2025-11-22T09:52:03.318108Z","iopub.status.idle":"2025-11-22T09:52:29.721074Z","shell.execute_reply.started":"2025-11-22T09:52:03.318088Z","shell.execute_reply":"2025-11-22T09:52:29.720193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom PIL import Image\n\n# --- CONFIGURATION ---\nSUBMISSION_FILE = 'submission.csv'\nIMAGE_DIR = '/kaggle/input/helipad-detection-challenge-sup-com/helipad_hackathon/images'\n\n# 1. LOAD RESULTS\nif not os.path.exists(SUBMISSION_FILE):\n    print(\"Error: submission.csv not found! Did you run the prediction step?\")\nelse:\n    df = pd.read_csv(SUBMISSION_FILE)\n    print(f\"Loaded predictions for {len(df)} images.\")\n\n    # 2. SHOW STATISTICS\n    count_0 = df[df['label'] == 0].shape[0]\n    count_1 = df[df['label'] == 1].shape[0]\n\n    print(f\"\\n--- PREDICTION STATS ---\")\n    print(f\"Predicted NO Helipad (0): {count_0}\")\n    print(f\"Predicted Helipad (1):    {count_1}\")\n    print(f\"Helipad Ratio: {count_1 / len(df) * 100:.2f}%\")\n\n    plt.figure(figsize=(6, 4))\n    sns.countplot(x=df['label'])\n    plt.title(\"Distribution of Predictions\")\n    plt.show()\n\n    # 3. VISUALIZE ACTUAL IMAGES\n    # Helper function to find image paths safely\n    def get_image_path(filename, root_dir):\n        path = os.path.join(root_dir, str(filename))\n        if os.path.exists(path): return path\n        if os.path.exists(path + '.jpg'): return path + '.jpg'\n        if os.path.exists(path + '.png'): return path + '.png'\n        return None\n\n    def show_predictions(label_value, title):\n        # Get random samples for this class\n        subset = df[df['label'] == label_value]\n        if len(subset) == 0:\n            print(f\"No images predicted as {title}\")\n            return\n        \n        samples = subset.sample(min(5, len(subset))) # Show up to 5\n        \n        plt.figure(figsize=(15, 5))\n        plt.suptitle(f\"Model Prediction: {title}\", fontsize=16)\n        \n        for i, (_, row) in enumerate(samples.iterrows()):\n            img_id = row['id'] # Adjust column name if different (e.g. 'ImageID')\n            img_path = get_image_path(img_id, IMAGE_DIR)\n            \n            plt.subplot(1, 5, i+1)\n            if img_path:\n                img = Image.open(img_path)\n                plt.imshow(img)\n                plt.title(img_id)\n            else:\n                plt.text(0.5, 0.5, \"Img Not Found\", ha='center')\n            plt.axis('off')\n        plt.show()\n\n    # Show images the model thinks are Helipads\n    show_predictions(1, \"Helipad (1)\")\n\n    # Show images the model thinks are Empty\n    show_predictions(0, \"No Helipad (0)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-22T09:55:16.125249Z","iopub.execute_input":"2025-11-22T09:55:16.125466Z","iopub.status.idle":"2025-11-22T09:55:17.430643Z","shell.execute_reply.started":"2025-11-22T09:55:16.125450Z","shell.execute_reply":"2025-11-22T09:55:17.429821Z"}},"outputs":[],"execution_count":null}]}