{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\ncount=0\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        count+=1\n        if count>20:\n            break\n\n    if count>20:\n        break\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:27.454645Z","iopub.execute_input":"2026-07-19T01:06:27.455725Z","iopub.status.idle":"2026-07-19T01:06:31.133669Z","shell.execute_reply.started":"2026-07-19T01:06:27.455679Z","shell.execute_reply":"2026-07-19T01:06:31.132891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom glob import glob\nimport cv2\nimport numpy as np\nimport torch\n\nbase_dir='../input/competitions/tensorflow-great-barrier-reef/train_images/'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.135146Z","iopub.execute_input":"2026-07-19T01:06:31.135397Z","iopub.status.idle":"2026-07-19T01:06:31.139488Z","shell.execute_reply.started":"2026-07-19T01:06:31.135374Z","shell.execute_reply":"2026-07-19T01:06:31.138805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df=pd.read_csv('/kaggle/input/competitions/tensorflow-great-barrier-reef/train.csv')\n\n#문자열을 리스트로 바꾸기(eval 함수)\ntrain_df['annotations'] =train_df['annotations'].apply(eval)\ntrain_df['image_path'] = train_df.apply(lambda x: f\"video_{x['video_id']}/{x['video_frame']}.jpg\", axis=1)\n\ntrain_df.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.140426Z","iopub.execute_input":"2026-07-19T01:06:31.140673Z","iopub.status.idle":"2026-07-19T01:06:31.654819Z","shell.execute_reply.started":"2026-07-19T01:06:31.140652Z","shell.execute_reply":"2026-07-19T01:06:31.654183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = cv2.imread(f\"{base_dir}video_0/0.jpg\")\nprint(img.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.656984Z","iopub.execute_input":"2026-07-19T01:06:31.657264Z","iopub.status.idle":"2026-07-19T01:06:31.677074Z","shell.execute_reply.started":"2026-07-19T01:06:31.657230Z","shell.execute_reply":"2026-07-19T01:06:31.676340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#전체 프레임 수의 90% 지점을 기준으로 앞쪽은 Train, 뒤쪽은 Val로 쪼개기\n#복잡한 서브시퀀스 계산 없이, 비디오 ID 내에서 단순 시간(프레임 순서)으로 가르기\n#이러면 약간의 data leakage가 발생한다. 일단 베이스라인 코드는 이렇게 작성\n\ntrain_df_list = []\nval_df_list = []\n\nfor video_id in [0, 1, 2]:\n    video_data = train_df[train_df['video_id'] == video_id].sort_values('video_frame')\n    split_idx = int(len(video_data) * 0.9)\n    \n    train_df_list.append(video_data.iloc[:split_idx])\n    val_df_list.append(video_data.iloc[split_idx:])\n\n# 다시 하나로 합치기\ntrain_df = pd.concat(train_df_list).reset_index(drop=True)\nval_df = pd.concat(val_df_list).reset_index(drop=True)\n\ntrain_df.shape[0], val_df.shape[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.678041Z","iopub.execute_input":"2026-07-19T01:06:31.678314Z","iopub.status.idle":"2026-07-19T01:06:31.704527Z","shell.execute_reply.started":"2026-07-19T01:06:31.678283Z","shell.execute_reply":"2026-07-19T01:06:31.703956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\ntrain_transform=A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.5),\n    ToTensorV2()\n], bbox_params=A.BboxParams(format='pascal_voc', label_fields=['labels']))   #label_field에 등록해놓으면 bbox랑 같이 필터링해준다.pascal_voc는 절대좌표 입력\n\n#faster rcnn은 resize도 모델 내부에서\n\ntest_transform=A.Compose([\n    ToTensorV2()\n], bbox_params=A.BboxParams(format='pascal_voc', label_fields=['labels']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.705543Z","iopub.execute_input":"2026-07-19T01:06:31.705895Z","iopub.status.idle":"2026-07-19T01:06:31.713404Z","shell.execute_reply.started":"2026-07-19T01:06:31.705869Z","shell.execute_reply":"2026-07-19T01:06:31.712591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import Dataset\n\nclass ReefDataset(Dataset):\n    def __init__(self, df, transform=None):\n        super().__init__()\n        self.df=df\n        self.transform=transform\n\n    def get_boxes(self, row):\n\n        if len(row['annotations']) == 0:\n            # 박스가 아예 없는 경우 -> 빈 (0, 4) 배열로 명시적 처리\n            return np.zeros((0, 4), dtype=np.float32)\n            \n        boxes = np.array([[a['x'], a['y'], a['width'], a['height']] for a in row['annotations']], dtype=np.float32)\n         \n        if boxes.ndim == 1:\n            boxes = boxes.reshape(1, -1)\n            \n        #[x_min, y_min, w, h]을 [x_min, y_min, x_max, y_max]로 바꾸기\n        boxes[:,2]=boxes[:,0]+boxes[:,2]\n        boxes[:,3]=boxes[:,1]+boxes[:,3]\n        return boxes\n\n    #증강할때 bbox 크가가 이미지 크기 넘어가면 에러. 미리 확인해주기\n    def can_augment(self, boxes):\n        box_outside_image = ((boxes[:, 0] < 0).any() or (boxes[:, 1] < 0).any() \n                             or (boxes[:, 2] > 1280).any() or (boxes[:, 3] > 720).any())\n        \n        return not box_outside_image\n\n    def get_image(self, row):\n        img=cv2.imread(f'{base_dir}{row['image_path']}', cv2.IMREAD_COLOR)  #IMREAD_COLOR은 강제 3채널 변경\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB).astype(np.float32)  #cv2.imread는 unit8로 반환. 정규화를 위해 float32로.\n        img/=255.0  #torchvision의 rcnn은 내부에서 imagenet mean/std로 정규화. 따라서 미리 0~1사이 픽셀값 가지도록 하기.\n        return img\n\n    def __getitem__(self, idx):\n        row=self.df.iloc[idx]\n        image=self.get_image(row)\n        boxes=self.get_boxes(row)\n\n        n_boxes=boxes.shape[0]\n\n        #target인자 중 area 계산하기. loss계산이나 순전파에는 사용 안되지만 coco eval이나 iscrowd영역 eval에서 제외할 때 사용\n        area=(boxes[:,2]-boxes[:,0])*(boxes[:,3]-boxes[:,1])\n\n        target={\n            'image_id': torch.tensor([idx], dtype=torch.int64),  #스칼라 텐서가 아닌 1차원 텐서를 rcnn이 받음.그래서 []사용\n            'boxes': torch.as_tensor(boxes, dtype=torch.float32),  #as_tensor로 복사 없이 기존 메모리를 그대로 참조\n            'labels': torch.ones((n_boxes,), dtype=torch.int64),  #만드는 텐서의 모양을 튜플로 받음\n            'area': torch.as_tensor(area, dtype=torch.float32),   #나중에 eval할때 크기별로 평가하기 위해 사용. 학습에 연관 없음.\n            'is_crowd': torch.zeros((n_boxes,), dtype=torch.int64)  #is_crowd인 area없음. is_crowd영역은 eval에서 특별취급\n        }\n        \n\n        if self.transform and self.can_augment(boxes):\n            sample = {\n                'image': image,\n                'bboxes': target['boxes'].numpy(),\n                'labels': target['labels'].numpy()\n            }\n            sample = self.transform(**sample)\n            image = sample['image']\n\n            if n_boxes>0:\n                target['boxes'] = torch.tensor(sample['bboxes'], dtype=torch.float32)   #sample의 boxes는 튜플 개수만큼의 리스트.텐서로 바꿔주기\n\n        else:\n            image=ToTensorV2()(image=image)['image']\n\n        return image, target\n\n\n    def __len__(self):\n        return len(self.df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.714507Z","iopub.execute_input":"2026-07-19T01:06:31.714869Z","iopub.status.idle":"2026-07-19T01:06:31.728218Z","shell.execute_reply.started":"2026-07-19T01:06:31.714833Z","shell.execute_reply":"2026-07-19T01:06:31.727586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset=ReefDataset(train_df, train_transform)\nval_dataset=ReefDataset(val_df, test_transform)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.729185Z","iopub.execute_input":"2026-07-19T01:06:31.729612Z","iopub.status.idle":"2026-07-19T01:06:31.741861Z","shell.execute_reply.started":"2026-07-19T01:06:31.729589Z","shell.execute_reply":"2026-07-19T01:06:31.741140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[train_df['annotations'].str.len()>10].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.742630Z","iopub.execute_input":"2026-07-19T01:06:31.743192Z","iopub.status.idle":"2026-07-19T01:06:31.777410Z","shell.execute_reply.started":"2026-07-19T01:06:31.743168Z","shell.execute_reply":"2026-07-19T01:06:31.776817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image, target=train_dataset[8621]\ntarget","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.779446Z","iopub.execute_input":"2026-07-19T01:06:31.779650Z","iopub.status.idle":"2026-07-19T01:06:31.826502Z","shell.execute_reply.started":"2026-07-19T01:06:31.779630Z","shell.execute_reply":"2026-07-19T01:06:31.825655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nboxes = target['boxes'].cpu().numpy().astype(np.int32)\nimg = image.permute(1,2,0).cpu().numpy()  #이미지가 텐서 타입이므로 순서 바꿔주기\nfig, ax = plt.subplots(1, 1, figsize=(16, 8))\n\nfor box in boxes:\n    cv2.rectangle(img,\n                  (box[0], box[1]),\n                  (box[2], box[3]),\n                  (220, 0, 0), 3)\n    \nax.set_axis_off()\nax.imshow(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:31.827487Z","iopub.execute_input":"2026-07-19T01:06:31.827807Z","iopub.status.idle":"2026-07-19T01:06:32.411639Z","shell.execute_reply.started":"2026-07-19T01:06:31.827773Z","shell.execute_reply":"2026-07-19T01:06:32.410383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport torch\n\ndef set_seed(seed=42):\n    random.seed(seed)    #파이선 기본 내장 random 모듈 시드\n    np.random.seed(seed)    #numpy 랜덤 시드, albumentation가 numpy랜덤 모듈을 사용\n    torch.manual_seed(seed)    #CPU에서 동작하는 pytorch 랜덤 시드(모델 가중치 초기화 등)\n    torch.cuda.manual_seed(seed)       # 단일 GPU용\n    torch.cuda.manual_seed_all(seed)    #멀티 GPU에서 동작하는 pytorch 랜덤 시드(dropout 등)\n\n\n#worker init fn(서브 프로세스 시드 고정)\ndef seed_worker(worker_id):\n    worker_seed = torch.initial_seed() % 2**32\n    np.random.seed(worker_seed)\n    random.seed(worker_seed)\n\ng = torch.Generator()  # 제너레이터 생성\ng.manual_seed(0)  # 제너레이터 시드값 고정\n#전역 시드 고정해도 프로세스들의 순서가 달라지면 학습되는 데이터 순서가 달라질 수 있으므로 shuffle을 통제\n\nset_seed()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:32.412723Z","iopub.execute_input":"2026-07-19T01:06:32.413088Z","iopub.status.idle":"2026-07-19T01:06:32.420101Z","shell.execute_reply.started":"2026-07-19T01:06:32.413063Z","shell.execute_reply":"2026-07-19T01:06:32.419344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import DataLoader\n\ndef collator(batch):\n    return tuple(zip(*batch))\n\ntrain_dataloader=DataLoader(train_dataset, batch_size=4, shuffle=True, drop_last=True, \n                            num_workers=4, worker_init_fn=seed_worker, generator=g, collate_fn=collator)\nval_dataloader=DataLoader(val_dataset, batch_size=4, shuffle=False, drop_last=False, \n                          num_workers=4, worker_init_fn=seed_worker, generator=g, collate_fn=collator)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:32.421440Z","iopub.execute_input":"2026-07-19T01:06:32.421700Z","iopub.status.idle":"2026-07-19T01:06:32.435881Z","shell.execute_reply.started":"2026-07-19T01:06:32.421678Z","shell.execute_reply":"2026-07-19T01:06:32.435073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from torchvision import models\nfrom torchvision import ops\nfrom torchvision.models.detection import rpn\nfrom torchvision.models.detection import FasterRCNN\n\ndevice='cuda' if torch.cuda.is_available() else 'cpu'\n\nbackbone=models.vgg16(weights='VGG16_Weights.IMAGENET1K_V1').features\nbackbone.out_channels=512 #faster rcnn 앵커에서 참조함\n\nanchor_generator=rpn.AnchorGenerator(\n    sizes=((32,64,128),),\n    aspect_ratios=((0.5,1.0,2.0,))\n)\n\nmodel=FasterRCNN(\n    backbone=backbone,\n    num_classes=2,\n    rpn_anchor_generator=anchor_generator,\n    min_size=480,  \n    max_size=800\n).to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:32.436880Z","iopub.execute_input":"2026-07-19T01:06:32.437197Z","iopub.status.idle":"2026-07-19T01:06:34.090236Z","shell.execute_reply.started":"2026-07-19T01:06:32.437174Z","shell.execute_reply":"2026-07-19T01:06:34.089622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch import optim\nfrom torch import nn\n\ngrad_params=[p for p in model.named_parameters() if p[1].requires_grad]\n\nno_decay = [\"bias\", \"norm\"]   # Bias나 LayerNorm은 weight decay 사용 안하고 보호(bias는 과적합에 영향x, layernorm은 모델 표현력을 올려주는 계층)\n\noptimizer_grouped_parameters = [\n    {\n        \"params\": [p for n, p in grad_params if not any(nd in n for nd in no_decay)],\n        \"weight_decay\": 1e-2,  #weight decay가 그래디언트 항하고 분리, lr과 곱해져서 빼지기 때문에 큰 값 아님.그리고 vit는 과적합에 취약\n    },\n    {\n        \"params\": [p for n, p in grad_params if any(nd in n for nd in no_decay)],\n        \"weight_decay\": 0.0, \n    },\n]\n\noptimizer = optim.AdamW(optimizer_grouped_parameters, lr=1e-3)\n\n#스케쥴러 설정-0.5배씩 줄이기\nscheduler=optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:34.091268Z","iopub.execute_input":"2026-07-19T01:06:34.092143Z","iopub.status.idle":"2026-07-19T01:06:34.099189Z","shell.execute_reply.started":"2026-07-19T01:06:34.092105Z","shell.execute_reply":"2026-07-19T01:06:34.098458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm.notebook import tqdm\n\nbest_val_loss=float('inf')\nsave_path='best_cots_model.pt'\nfor epoch in range(5):\n    \n    #모델 훈련\n    model.train()\n    train_loss=0.0\n\n    for images, targets in tqdm(train_dataloader):\n        images=list(image.to(device) for image in images)\n        targets=[{k:v.to(device) for k,v in t.items()} for t in targets]\n\n        loss_dict=model(images, targets)\n        losses=sum(loss for loss in loss_dict.values())\n        \n        optimizer.zero_grad()\n        losses.backward()   #loss dict에 있는 값들이 다 낮아져야 하므로 loss를 다 더해서 역전파 진행\n        optimizer.step()\n\n        train_loss+=losses.item()\n\n\n    train_loss=train_loss/len(train_dataloader)\n    print(f'Epoch: {epoch+1:4d}, Train Loss: {train_loss:.4f}')\n\n    model.train()  #faster rcnn은 eval 모드에서 loss를 반환하지 않음. 그래서 val에서도 train 모\n    val_loss=0\n\n    with torch.no_grad():\n        for images, targets in tqdm(val_dataloader):\n            images=list(image.to(device) for image in images)\n            targets=[{k:v.to(device) for k,v in t.items()} for t in targets]\n\n            loss_dict=model(images, targets)\n            losses=sum(loss for loss in loss_dict.values())\n\n            val_loss+=losses.item()*len(images)\n\n    val_loss=val_loss/len(val_dataloader.dataset)\n\n    scheduler.step(val_loss)\n    current_lr=optimizer.param_groups[0]['lr']\n    print(f'Validation Loss: {val_loss:.4f}, Current LR: {current_lr:.6f}')\n\n    if val_loss<best_val_loss:\n        print('가장 낮은 val loss 기록, 모델 저장')\n\n        best_val_loss=val_loss\n        torch.save(model.state_dict(), save_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-19T01:06:34.100158Z","iopub.execute_input":"2026-07-19T01:06:34.100455Z","iopub.status.idle":"2026-07-19T03:26:20.413058Z","shell.execute_reply.started":"2026-07-19T01:06:34.100431Z","shell.execute_reply":"2026-07-19T03:26:20.411996Z"}},"outputs":[],"execution_count":null}]}