{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":98450,"databundleVersionId":11749951,"sourceType":"competition"},{"sourceId":239995795,"sourceType":"kernelVersion"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Spectrum 32 Lightning 3DCNN\n","metadata":{"papermill":{"duration":0.003895,"end_time":"2023-06-30T09:39:19.33539","exception":false,"start_time":"2023-06-30T09:39:19.331495","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install lightning ","metadata":{"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:20:13.200316Z","iopub.execute_input":"2025-05-18T03:20:13.201223Z","iopub.status.idle":"2025-05-18T03:21:39.63955Z","shell.execute_reply.started":"2025-05-18T03:20:13.201192Z","shell.execute_reply":"2025-05-18T03:21:39.637675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import random_split\nfrom torch.utils.data import DataLoader, Dataset, Subset\nfrom torch.utils.data import random_split, SubsetRandomSampler\nfrom torchvision import datasets, transforms, models \nfrom torchvision.datasets import ImageFolder\nfrom torchvision.transforms import ToTensor\nfrom torchvision.utils import make_grid\n\n#the latest environement\nimport lightning.pytorch as L\nfrom lightning.pytorch import LightningDataModule\nfrom lightning.pytorch import LightningModule\nfrom lightning.pytorch import Trainer\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom PIL import Image","metadata":{"papermill":{"duration":13.262195,"end_time":"2023-06-30T09:39:32.607264","exception":false,"start_time":"2023-06-30T09:39:19.345069","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:21:39.642363Z","iopub.execute_input":"2025-05-18T03:21:39.642721Z","iopub.status.idle":"2025-05-18T03:21:48.712373Z","shell.execute_reply.started":"2025-05-18T03:21:39.642685Z","shell.execute_reply":"2025-05-18T03:21:48.711165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dir0='/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/ot/ot'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:21:48.721212Z","iopub.execute_input":"2025-05-18T03:21:48.721479Z","iopub.status.idle":"2025-05-18T03:21:48.746208Z","shell.execute_reply.started":"2025-05-18T03:21:48.721452Z","shell.execute_reply":"2025-05-18T03:21:48.745241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data057=pd.read_csv('/kaggle/input/visible-spectrum-split-data-by-shape/data057.csv')\ndata128=pd.read_csv('/kaggle/input/visible-spectrum-split-data-by-shape/data128.csv')\ndata=pd.concat([data057,data128],axis=0)\ntrain=data[data['traintest']=='train']\nTEST=data[data['traintest']=='test']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:21:48.747531Z","iopub.execute_input":"2025-05-18T03:21:48.747829Z","iopub.status.idle":"2025-05-18T03:21:48.82362Z","shell.execute_reply.started":"2025-05-18T03:21:48.747804Z","shell.execute_reply":"2025-05-18T03:21:48.822711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(TEST[0:2].T)\nTEST['label']=0 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:21:48.824764Z","iopub.execute_input":"2025-05-18T03:21:48.825089Z","iopub.status.idle":"2025-05-18T03:21:48.853115Z","shell.execute_reply.started":"2025-05-18T03:21:48.825054Z","shell.execute_reply":"2025-05-18T03:21:48.852374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_names = [str(i) for i in range(101)] ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.40238Z","iopub.status.idle":"2025-05-18T03:10:00.403424Z","shell.execute_reply.started":"2025-05-18T03:10:00.40307Z","shell.execute_reply":"2025-05-18T03:10:00.403086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_path_label_list(df):\n    path_label_list = []\n    for _, row in df.iterrows():\n        path = row['path']\n        label = row['label']\n        path_label_list.append((path, label))\n    return path_label_list\n\npath_label = create_path_label_list(train)\nprint(path_label[0:3])","metadata":{"papermill":{"duration":0.299275,"end_time":"2023-06-30T09:39:32.987661","exception":false,"start_time":"2023-06-30T09:39:32.688386","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.404478Z","iopub.status.idle":"2025-05-18T03:10:00.404859Z","shell.execute_reply.started":"2025-05-18T03:10:00.404674Z","shell.execute_reply":"2025-05-18T03:10:00.404693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TEST_create_path_label_list(df):\n    path_label_list = []\n    for _, row in df.iterrows():\n        path= row['path']\n        label = row['label']\n        path_label_list.append((path, label))\n    return path_label_list\n\nTEST_path_label = TEST_create_path_label_list(TEST)\nprint(path_label[0:3])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.40604Z","iopub.status.idle":"2025-05-18T03:10:00.406397Z","shell.execute_reply.started":"2025-05-18T03:10:00.40622Z","shell.execute_reply":"2025-05-18T03:10:00.406237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CustomDataset(torch.utils.data.Dataset):\n    def __init__(self, path_label, transform=None):\n        # path_label should be a list of tuples containing (data_path, label)\n        self.path_label = path_label\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.path_label)\n\n    def __getitem__(self, idx):\n        # Load data and get label\n        rgb = np.load(self.path_label[idx][0])  # Load data from path\n        rgb = rgb.astype('uint8')  # Or torch.FloatTensor\n        label = self.path_label[idx][1]  # Get the label\n        \n        if self.transform:\n            rgb = self.transform(rgb)\n\n        return rgb, torch.tensor(label, dtype=torch.long) \n        \n###------------------------------------------------------------------------","metadata":{"papermill":{"duration":0.014251,"end_time":"2023-06-30T09:39:33.032042","exception":false,"start_time":"2023-06-30T09:39:33.017791","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.409591Z","iopub.status.idle":"2025-05-18T03:10:00.409979Z","shell.execute_reply.started":"2025-05-18T03:10:00.409771Z","shell.execute_reply":"2025-05-18T03:10:00.409787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Resize3D:\n    def __init__(self, size):  # size = (D_new, H_new, W_new)\n        self.size = size\n\n    def __call__(self, img):\n        # Check if the input is a NumPy array or a tensor\n        if isinstance(img, np.ndarray):\n            img_tensor = torch.from_numpy(img).float()\n        elif isinstance(img, torch.Tensor):\n            img_tensor = img.float()\n        else:\n            raise TypeError(\"Input must be either numpy.ndarray or torch.Tensor\")\n        \n        # Convert shape from (D, H, W) to (1, 1, D, H, W)\n        img_tensor = img_tensor.unsqueeze(0).unsqueeze(0)\n        \n        # Apply 3D resizing\n        resized = F.interpolate(\n            img_tensor,\n            size=self.size,\n            mode='trilinear',\n            align_corners=False\n        )\n        \n        # Convert shape back from (1, 1, D_new, H_new, W_new) to (D_new, H_new, W_new)\n        return resized.squeeze(0).squeeze(0)\n\n###------------------------------------------------------------------------","metadata":{"papermill":{"duration":0.014251,"end_time":"2023-06-30T09:39:33.032042","exception":false,"start_time":"2023-06-30T09:39:33.017791","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.409591Z","iopub.status.idle":"2025-05-18T03:10:00.409979Z","shell.execute_reply.started":"2025-05-18T03:10:00.409771Z","shell.execute_reply":"2025-05-18T03:10:00.409787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DataModule(LightningDataModule):\n    def __init__(self, path_label, batch_size=1, test_path_label=None, num_workers=4):\n        super().__init__()\n        self.path_label = path_label\n        self.test_path_label = test_path_label\n        self.batch_size = batch_size\n        self.num_workers = num_workers\n        \n        # Transformations for 3D data (example normalization for grayscale medical images)\n        self.transform = transforms.Compose([\n            transforms.ToTensor(),\n            Resize3D((32, 32, 32)),  # (Depth, Height, Width)(125, 128, 128)\n            transforms.Normalize(mean=[0.5], std=[0.5])  # For single-channel data\n        ])\n\n    def setup(self, stage=None):\n        if stage == 'test':\n            self.test_dataset = CustomDataset(self.test_path_label, self.transform)\n            \n        elif stage == 'fit' or stage is None:\n            full_dataset = CustomDataset(self.path_label, self.transform)\n            \n            # Random split (shuffling is recommended)\n            train_size = int(0.8 * len(full_dataset))\n            val_size = len(full_dataset) - train_size\n            self.train_dataset, self.val_dataset = random_split(\n                full_dataset, \n                [train_size, val_size],\n                generator=torch.Generator().manual_seed(42)  # Ensures reproducibility\n            )\n\n    def train_dataloader(self):\n        return DataLoader(\n            self.train_dataset,\n            batch_size=self.batch_size,\n            shuffle=True,\n            num_workers=self.num_workers,\n            pin_memory=True  # Enable when using GPU\n        )\n\n    def val_dataloader(self):\n        return DataLoader(\n            self.val_dataset,\n            batch_size=self.batch_size,\n            num_workers=self.num_workers\n        )\n\n    def test_dataloader(self):\n        return DataLoader(\n            self.test_dataset,\n            batch_size=self.batch_size,\n            num_workers=self.num_workers\n        )\n\n    def configure_optimizers(self):\n        optimizer = torch.optim.Adam(self.parameters(), lr=0.001)\n        return optimizer\n\n\n        ","metadata":{"papermill":{"duration":0.014251,"end_time":"2023-06-30T09:39:33.032042","exception":false,"start_time":"2023-06-30T09:39:33.017791","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.409591Z","iopub.status.idle":"2025-05-18T03:10:00.409979Z","shell.execute_reply.started":"2025-05-18T03:10:00.409771Z","shell.execute_reply":"2025-05-18T03:10:00.409787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Conv3DNetwork(LightningModule):\n    def __init__(self, in_channels=1, num_classes=101):\n        super().__init__()\n        self.conv_layers = nn.Sequential(\n            nn.Conv3d(in_channels, 16, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool3d(2),  # (1,32,32,32) → (16,16,16,16)\n            nn.Conv3d(16, 32, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool3d(2)   # (16,16,16,16) → (32,8,8,8)\n        )\n        self.gap = nn.AdaptiveAvgPool3d(1)  # (32,1,1,1)\n        self.fc = nn.Linear(32, num_classes)  # Input has 32 channels\n\n    def forward(self, x):\n        # Check input shape (batch, channels, depth, height, width)\n        if x.dim() == 4:  # If the channel dimension is missing\n            x = x.unsqueeze(1)  # (b,d,h,w) → (b,1,d,h,w)\n        \n        x = self.conv_layers(x)\n        #print(f\"Conv out: {x.shape}\")  # Should be [batch, 32, 8, 8, 8]\n        x = self.gap(x)\n        x = x.view(x.size(0), x.size(1))  # [batch, channels]\n        return F.log_softmax(self.fc(x), dim=1)\n\n\n\n\n\n\n    \n    def configure_optimizers(self):\n        optimizer = torch.optim.Adam(self.parameters(), lr=0.001)\n        return optimizer\n\n    def training_step(self, train_batch, batch_idx):\n        try:\n            X, y = train_batch[0], train_batch[1]  # Get the first two elements\n        except (IndexError, TypeError) as e:\n            raise ValueError(f\"Unexpected batch format: {type(train_batch)}\") from e\n        \n        y_hat = self(X)  # Forward pass\n        loss = F.cross_entropy(y_hat, y)  # Compute loss\n        pred = y_hat.argmax(dim=1, keepdim=True)  # Get predicted classes\n        acc = pred.eq(y.view_as(pred)).sum().item() / y.shape[0]  # Compute accuracy\n        self.log(\"train_loss\", loss, prog_bar=True)\n        self.log(\"train_acc\", acc, prog_bar=True)\n        return loss\n    \n    def validation_step(self, val_batch, batch_idx):\n        X, y = val_batch[0], val_batch[1]  # Explicit indexing\n        y_hat = self(X)\n        loss = F.cross_entropy(y_hat, y)\n        pred = y_hat.argmax(dim=1, keepdim=True)\n        acc = pred.eq(y.view_as(pred)).sum().item() / y.shape[0]\n        self.log(\"val_loss\", loss, prog_bar=True)\n        self.log(\"val_acc\", acc, prog_bar=True)\n    \n    def test_step(self, test_batch, batch_idx):\n        X, y = test_batch[0], test_batch[1]\n        y_hat = self(X)\n        loss = F.cross_entropy(y_hat, y)\n        pred = y_hat.argmax(dim=1, keepdim=True)\n        acc = pred.eq(y.view_as(pred)).sum().item() / y.shape[0]\n        self.log(\"test_loss\", loss)\n        self.log(\"test_acc\", acc)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datamodule = DataModule(path_label=path_label,test_path_label=TEST_path_label)\ndatamodule.setup(stage=None)\n\nmodel = Conv3DNetwork()\ntrainer = L.Trainer(max_epochs=200)\ntrainer.fit(model, datamodule)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.413731Z","iopub.status.idle":"2025-05-18T03:10:00.414003Z","shell.execute_reply.started":"2025-05-18T03:10:00.413855Z","shell.execute_reply":"2025-05-18T03:10:00.413866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_loader = datamodule.val_dataloader()\ntrainer.validate(dataloaders=val_loader)","metadata":{"papermill":{"duration":959.542695,"end_time":"2023-06-30T09:55:32.61869","exception":false,"start_time":"2023-06-30T09:39:33.075995","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.416888Z","iopub.status.idle":"2025-05-18T03:10:00.417202Z","shell.execute_reply.started":"2025-05-18T03:10:00.417078Z","shell.execute_reply":"2025-05-18T03:10:00.417092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cpu\")   #\"cuda:0\"\n\nmodel.eval()\ny_true=[]\ny_pred=[]\nwith torch.no_grad():\n    for val_data in datamodule.val_dataloader():\n        val_images, val_labels = val_data[0].to(device), val_data[1].to(device)\n        pred = model(val_images).argmax(dim=1)\n        for i in range(len(pred)):\n            y_true.append(val_labels[i].item())\n            y_pred.append(pred[i].item())\n\nclass_names = [str(i) for i in range(101)] \nprint(classification_report(y_true, y_pred, target_names=class_names, labels=list(range(101)), digits=4))","metadata":{"papermill":{"duration":5.368379,"end_time":"2023-06-30T09:55:39.113208","exception":false,"start_time":"2023-06-30T09:55:33.744829","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2025-05-18T03:10:00.4201Z","iopub.status.idle":"2025-05-18T03:10:00.420444Z","shell.execute_reply.started":"2025-05-18T03:10:00.420233Z","shell.execute_reply":"2025-05-18T03:10:00.420244Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datamodule.setup(stage='test')","metadata":{"papermill":{"duration":0.008862,"end_time":"2023-06-30T09:55:39.131596","exception":false,"start_time":"2023-06-30T09:55:39.122734","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.422102Z","iopub.status.idle":"2025-05-18T03:10:00.422403Z","shell.execute_reply.started":"2025-05-18T03:10:00.422275Z","shell.execute_reply":"2025-05-18T03:10:00.422291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\ny_true=[]\ny_pred=[]\nwith torch.no_grad():\n    for test_data in datamodule.test_dataloader():\n        test_images, test_labels =test_data[0].to(device), test_data[1].to(device)\n        pred = model(test_images).argmax(dim=1)\n        for i in range(len(pred)):\n            y_true.append(test_labels[i].item())\n            y_pred.append(pred[i].item())","metadata":{"papermill":{"duration":0.009079,"end_time":"2023-06-30T09:55:39.150422","exception":false,"start_time":"2023-06-30T09:55:39.141343","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.423641Z","iopub.status.idle":"2025-05-18T03:10:00.423973Z","shell.execute_reply.started":"2025-05-18T03:10:00.423812Z","shell.execute_reply":"2025-05-18T03:10:00.423827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST['label']=y_pred\ndisplay(TEST)\n\nsubmit=TEST[['id','label']]\nsubmit.columns=['ID','TARGET']\nsubmit.to_csv('submission.csv',index=False)\ndisplay(submit)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T03:10:00.42528Z","iopub.status.idle":"2025-05-18T03:10:00.425582Z","shell.execute_reply.started":"2025-05-18T03:10:00.42542Z","shell.execute_reply":"2025-05-18T03:10:00.425434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}