{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":9988,"databundleVersionId":868324,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torchvision\nfrom torch.utils.data import Dataset, DataLoader\nimport pandas as pd\nfrom PIL import Image\nfrom torchvision.transforms import v2\nimport os\nfrom sklearn.model_selection import train_test_split\nimport random\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:13:53.975534Z","iopub.execute_input":"2025-03-24T19:13:53.975930Z","iopub.status.idle":"2025-03-24T19:13:53.981164Z","shell.execute_reply.started":"2025-03-24T19:13:53.975870Z","shell.execute_reply":"2025-03-24T19:13:53.980069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def set_seed(seed=None, seed_torch=True):\n    \"\"\"\n    Function that controls randomness. NumPy and random modules must be imported.\n\n    Args:\n      seed : Integer\n        A non-negative integer that defines the random state. Default is `None`.\n      seed_torch : Boolean\n        If `True` sets the random seed for pytorch tensors, so pytorch module\n        must be imported. Default is `True`.\n\n    Returns:\n      Nothing.\n    \"\"\"\n    if seed is None:\n        seed = np.random.choice(2 ** 32)\n    random.seed(seed)\n    np.random.seed(seed)\n    if seed_torch:\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n        torch.cuda.manual_seed(seed)\n        torch.backends.cudnn.benchmark = False\n        torch.backends.cudnn.deterministic = True\n    print(f'Random seed {seed} has been set.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:10:01.851733Z","iopub.execute_input":"2025-03-24T19:10:01.852349Z","iopub.status.idle":"2025-03-24T19:10:01.858421Z","shell.execute_reply.started":"2025-03-24T19:10:01.852307Z","shell.execute_reply":"2025-03-24T19:10:01.857296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_images_from_dataloader(dataloader):\n    for images, labels in dataloader:\n        # Plot batch of images\n        fig, axes = plt.subplots(1, len(images), figsize=(15, 5))\n        if len(images) == 1:\n            axes = [axes]  # Ensure it's iterable for a single image\n\n        for i, (image, label) in enumerate(zip(images, labels)):\n            # Convert the image back to a NumPy array for plotting\n            image = image.permute(1, 2, 0).numpy()  # Change from (C, H, W) to (H, W, C)\n            image = image * [0.229, 0.224, 0.225] + [0.485, 0.456, 0.406]  # Denormalize\n            image = image.clip(0, 1)  # Clip values to valid range [0, 1]\n\n            axes[i].imshow(image)\n            axes[i].set_title(f'Label: {label.item()}')\n            axes[i].axis('off')\n\n        plt.show()\n        break  # Display only the first batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:13:11.108848Z","iopub.execute_input":"2025-03-24T19:13:11.109236Z","iopub.status.idle":"2025-03-24T19:13:11.115655Z","shell.execute_reply.started":"2025-03-24T19:13:11.109206Z","shell.execute_reply":"2025-03-24T19:13:11.114633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DataProcessor:\n    def __init__(self, dataset_root: str, csv_file: str, seed: int):\n        self.dataset_root = dataset_root\n        self.csv_file = csv_file\n        self.seed = seed\n        self.csv_path = os.path.join(self.dataset_root, self.csv_file)\n        self.df = self.__preprocess_data()\n        self.train_data, self.val_data, self.test_data = self.__split_data()\n\n    def __preprocess_data(self) -> pd.DataFrame:\n        df = pd.read_csv(self.csv_path)\n        df['ImagePath'] = df['ImageId'].apply(lambda id: os.path.join(self.dataset_root, 'train_v2', id))\n        df['label'] = df['EncodedPixels'].notna().astype(int)\n        df.drop(columns=['ImageId', 'EncodedPixels'], inplace=True)\n        return df\n\n    def get_data(self) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:\n        return self.train_data, self.val_data, self.test_data\n\n    def __split_data(self) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:\n        train, evaluation = train_test_split(\n            self.df,\n            test_size=0.2,\n            random_state=self.seed,\n            stratify=self.df['label']\n        )\n        val, test = train_test_split(\n            evaluation,\n            test_size=0.5,\n            random_state=self.seed,\n            stratify=evaluation['label']\n        )\n        return train, val, test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:10:01.860593Z","iopub.execute_input":"2025-03-24T19:10:01.860893Z","iopub.status.idle":"2025-03-24T19:10:01.883388Z","shell.execute_reply.started":"2025-03-24T19:10:01.860847Z","shell.execute_reply":"2025-03-24T19:10:01.882111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ClassificationDataset(Dataset):\n    def __init__(self, df: pd.DataFrame, transform=None):\n        self.df = df\n        self.transform = transform or v2.Compose([\n            v2.Resize((224, 224)),\n            v2.ToImage(),\n            v2.ToDtype(torch.float32, scale=True),\n            v2.Normalize(mean=[0.485, 0.456, 0.406],std=[0.229, 0.224, 0.225])\n        ])\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        image_path = self.df.iloc[idx]['ImagePath']\n        label = self.df.iloc[idx]['label']\n\n        image = Image.open(image_path).convert('RGB')\n        image = self.transform(image)\n\n        label = torch.tensor(label, dtype=torch.long)\n\n        return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:11:18.322484Z","iopub.execute_input":"2025-03-24T19:11:18.322987Z","iopub.status.idle":"2025-03-24T19:11:18.330110Z","shell.execute_reply.started":"2025-03-24T19:11:18.322951Z","shell.execute_reply":"2025-03-24T19:11:18.328786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_root = '/kaggle/input/airbus-ship-detection'\ncsv_file = '/kaggle/input/airbus-ship-detection/train_ship_segmentations_v2.csv'\nseed = 2005\nset_seed(seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:13:36.726143Z","iopub.execute_input":"2025-03-24T19:13:36.726517Z","iopub.status.idle":"2025-03-24T19:13:36.734360Z","shell.execute_reply.started":"2025-03-24T19:13:36.726486Z","shell.execute_reply":"2025-03-24T19:13:36.733363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dp = DataProcessor(dataset_root, csv_file, seed)\ntrain, val, test = dp.get_data()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:13:39.281270Z","iopub.execute_input":"2025-03-24T19:13:39.281605Z","iopub.status.idle":"2025-03-24T19:13:40.501542Z","shell.execute_reply.started":"2025-03-24T19:13:39.281579Z","shell.execute_reply":"2025-03-24T19:13:40.500667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = ClassificationDataset(train)\ndataloader = DataLoader(dataset, batch_size=2, shuffle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:14:18.610320Z","iopub.execute_input":"2025-03-24T19:14:18.610704Z","iopub.status.idle":"2025-03-24T19:14:18.615960Z","shell.execute_reply.started":"2025-03-24T19:14:18.610672Z","shell.execute_reply":"2025-03-24T19:14:18.614638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_images_from_dataloader(dataloader)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T19:14:35.217162Z","iopub.execute_input":"2025-03-24T19:14:35.217534Z","iopub.status.idle":"2025-03-24T19:14:35.678005Z","shell.execute_reply.started":"2025-03-24T19:14:35.217507Z","shell.execute_reply":"2025-03-24T19:14:35.676631Z"}},"outputs":[],"execution_count":null}]}