{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":128792,"databundleVersionId":15494745}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom collections import Counter\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:32.597092Z","iopub.execute_input":"2026-02-23T15:13:32.597555Z","iopub.status.idle":"2026-02-23T15:13:32.604158Z","shell.execute_reply.started":"2026-02-23T15:13:32.597515Z","shell.execute_reply":"2026-02-23T15:13:32.603048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/vista26/Vistas Dataset Public/Vistas Dataset Public\"\nTRAIN_JSON = os.path.join(BASE_PATH, \"instances_train.json\")\nTEST_JSON = os.path.join(BASE_PATH, \"instances_test.json\")\nVAL_JSON = \"/kaggle/input/vista26/instances_val.json\"\n\nCATEGORY_JSON = os.path.join(BASE_PATH, \"Categories.json\")\n\nTRAIN_DIR = os.path.join(BASE_PATH, \"train\")\nTEST_DIR = os.path.join(BASE_PATH, \"test\")\nVAL_DIR = os.path.join(BASE_PATH, \"validation\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:32.606166Z","iopub.execute_input":"2026-02-23T15:13:32.606556Z","iopub.status.idle":"2026-02-23T15:13:32.621512Z","shell.execute_reply.started":"2026-02-23T15:13:32.606519Z","shell.execute_reply":"2026-02-23T15:13:32.620281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load annotation files\n\nwith open(TRAIN_JSON) as f:\n    train_data = json.load(f)\n\nwith open(TEST_JSON) as f:\n    test_data = json.load(f)\n\nwith open(VAL_JSON) as f:\n    val_data = json.load(f)\n\nwith open(CATEGORY_JSON) as f:\n    categories = json.load(f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:32.623049Z","iopub.execute_input":"2026-02-23T15:13:32.623358Z","iopub.status.idle":"2026-02-23T15:13:34.509228Z","shell.execute_reply.started":"2026-02-23T15:13:32.623331Z","shell.execute_reply":"2026-02-23T15:13:34.508187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(type(categories))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:34.511587Z","iopub.execute_input":"2026-02-23T15:13:34.511947Z","iopub.status.idle":"2026-02-23T15:13:34.51799Z","shell.execute_reply.started":"2026-02-23T15:13:34.511916Z","shell.execute_reply":"2026-02-23T15:13:34.516805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Bulid category mapping\n\ncat_id_to_name = {c[\"id\"]: c[\"name\"] for c in categories[\"categories\"]}\nprint(\"Total Categories: \", len(cat_id_to_name))\n\ncat_id_to_supercat = {\n    c[\"id\"]: c[\"supercategory\"]\n    for c in categories[\"categories\"]\n}\n\nprint(\"Example: \")\nprint(\"Category 1:\", cat_id_to_name[1])\nprint(\"Supercategory: \", cat_id_to_supercat[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:34.519182Z","iopub.execute_input":"2026-02-23T15:13:34.519765Z","iopub.status.idle":"2026-02-23T15:13:34.542109Z","shell.execute_reply.started":"2026-02-23T15:13:34.519705Z","shell.execute_reply":"2026-02-23T15:13:34.540637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#supercategory distribution\n\ncat_df = pd.DataFrame(categories[\"categories\"])\n\nsupercat_counts = cat_df[\"supercategory\"].value_counts()\n\nplt.figure(figsize=(10, 6))\nsupercat_counts.plot(kind=\"bar\")\nplt.title(\"SuperCategory Distribution\")\nplt.ylabel(\"Number of classes\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:34.543229Z","iopub.execute_input":"2026-02-23T15:13:34.543639Z","iopub.status.idle":"2026-02-23T15:13:35.525856Z","shell.execute_reply.started":"2026-02-23T15:13:34.543602Z","shell.execute_reply":"2026-02-23T15:13:35.524677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Basic Dataset Statistics\n\nprint(\"Train Images: \", len(train_data[\"images\"]))\nprint(\"Train Annotations: \", len(train_data[\"annotations\"]))\n\nprint(\"Test Images: \", len(test_data[\"images\"]))\nprint(\"Test Annotations: \", len(test_data[\"annotations\"]))\n\nprint(\"Validation Images: \", len(val_data[\"images\"]))\n#print(\"Validation Annotations: \", len(val_data[\"annotations\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:40:18.014346Z","iopub.execute_input":"2026-02-23T15:40:18.0152Z","iopub.status.idle":"2026-02-23T15:40:18.021545Z","shell.execute_reply.started":"2026-02-23T15:40:18.015164Z","shell.execute_reply":"2026-02-23T15:40:18.020337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(val_data.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:43.966425Z","iopub.execute_input":"2026-02-23T15:13:43.966847Z","iopub.status.idle":"2026-02-23T15:13:43.973579Z","shell.execute_reply.started":"2026-02-23T15:13:43.966815Z","shell.execute_reply":"2026-02-23T15:13:43.972114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#class distribution\n#Training class distribution\n\ntrain_cat_ids = [ann[\"category_id\"] for ann in train_data[\"annotations\"]]\ntrain_class_counts = Counter(train_cat_ids)\n\ntrain_class_df = pd.DataFrame({\n    'category_id' : train_class_counts.keys(),\n    \"count\" : train_class_counts.values()\n})\n\ntrain_class_df[\"category_name\"] = train_class_df[\"category_id\"].map(cat_id_to_name)\ntrain_class_df = train_class_df.sort_values(\"count\", ascending=False)\n\ntrain_class_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:45.028334Z","iopub.execute_input":"2026-02-23T15:13:45.028649Z","iopub.status.idle":"2026-02-23T15:13:45.072252Z","shell.execute_reply.started":"2026-02-23T15:13:45.028625Z","shell.execute_reply":"2026-02-23T15:13:45.071246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nplt.bar(train_class_df[\"category_id\"], train_class_df[\"count\"])\nplt.title(\"Training Class Distribution\")\nplt.xlabel(\"Category ID\")\nplt.ylabel(\"Count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:47.481121Z","iopub.execute_input":"2026-02-23T15:13:47.481455Z","iopub.status.idle":"2026-02-23T15:13:47.842074Z","shell.execute_reply.started":"2026-02-23T15:13:47.481426Z","shell.execute_reply":"2026-02-23T15:13:47.841156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#object per image analysis\n#Train\ntrain_img_object_count = Counter([ann[\"image_id\"] for ann in train_data[\"annotations\"]])\ntrain_obj_counts = list(train_img_object_count.values())\n\nprint(\"Train Avg objects per image: \", np.mean(train_obj_counts))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:50.372297Z","iopub.execute_input":"2026-02-23T15:13:50.373234Z","iopub.status.idle":"2026-02-23T15:13:50.397849Z","shell.execute_reply.started":"2026-02-23T15:13:50.373194Z","shell.execute_reply":"2026-02-23T15:13:50.396868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Test\ntest_img_object_count = Counter([ann[\"image_id\"] for ann in test_data[\"annotations\"]])\ntest_obj_counts = list(test_img_object_count.values())\n\nprint(\"Test Avg Objects per image: \", np.mean(test_obj_counts))\nprint(\"Max object in Test Images: \", max(test_obj_counts))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:13:55.671624Z","iopub.execute_input":"2026-02-23T15:13:55.671973Z","iopub.status.idle":"2026-02-23T15:13:55.739939Z","shell.execute_reply.started":"2026-02-23T15:13:55.671947Z","shell.execute_reply":"2026-02-23T15:13:55.738851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Distribution plot\n\nplt.hist(test_obj_counts, bins=20)\nplt.title(\"Test Objects Per Image Distribution\")\nplt.xlabel(\"Number of Objects\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:14:00.917491Z","iopub.execute_input":"2026-02-23T15:14:00.91785Z","iopub.status.idle":"2026-02-23T15:14:01.144023Z","shell.execute_reply.started":"2026-02-23T15:14:00.917822Z","shell.execute_reply":"2026-02-23T15:14:01.142886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Bounding box size analysis\n\nbbox_areas = []\n\nfor ann in train_data[\"annotations\"]:\n    x, y, w, h = ann[\"bbox\"]\n    bbox_areas.append(w * h)\n\nplt.hist(bbox_areas, bins=50)\nplt.title(\"Bounding Box Area Distribution (Train)\")\nplt.xlabel(\"Area\")\nplt.ylabel(\"Freqency\")\nplt.show()\n\n#print(\"Mean BBox Area: \", np.mean(bbox_areas))\n#print(\"Median BBos Area: \", np.median(bbox_areas))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T16:03:09.243079Z","iopub.execute_input":"2026-02-23T16:03:09.24343Z","iopub.status.idle":"2026-02-23T16:03:09.552894Z","shell.execute_reply.started":"2026-02-23T16:03:09.243405Z","shell.execute_reply":"2026-02-23T16:03:09.552014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#relative BBox size (Small vs Large objects)\n\nrelative_sizes = []\n\nfor ann in train_data[\"annotations\"]:\n    img = next(img for img in train_data[\"images\"] if img[\"id\"] == ann[\"image_id\"])\n    img_area = img[\"width\"] * img[\"height\"]\n    x, y, w, h = ann[\"bbox\"]\n    relative_sizes.append((w * h) / img_area)\n\nplt.hist(relative_sizes, bins=50)\nplt.title(\"Relative BBox Size Distribution\")\nplt.xlabel(\"BBox / Image Area\")\nplt.ylabel(\"Frequency\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:26:42.197271Z","iopub.execute_input":"2026-02-23T15:26:42.197794Z","iopub.status.idle":"2026-02-23T15:28:40.403034Z","shell.execute_reply.started":"2026-02-23T15:26:42.197731Z","shell.execute_reply":"2026-02-23T15:28:40.40199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Difficulty level Distribution (Test)\n\ndifficulty_counts = Counter([img.get(\"difficulty\", \"unknown\") for img in test_data[\"images\"]])\nprint(\"Test Difficulty Distribution: \", difficulty_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:33:17.056839Z","iopub.execute_input":"2026-02-23T15:33:17.057229Z","iopub.status.idle":"2026-02-23T15:33:17.066038Z","shell.execute_reply.started":"2026-02-23T15:33:17.057201Z","shell.execute_reply":"2026-02-23T15:33:17.064955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Image resolution Distribution\n\ntrain_widths = [img[\"width\"] for img in train_data[\"images\"]]\ntrain_heights = [img[\"height\"] for img in train_data[\"images\"]]\n\nplt.scatter(train_widths, train_heights)\nplt.title(\"Train Image Resolution Distribution\")\nplt.xlabel(\"Width\")\nplt.ylabel(\"Height\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:37:09.622951Z","iopub.execute_input":"2026-02-23T15:37:09.623335Z","iopub.status.idle":"2026-02-23T15:37:10.169069Z","shell.execute_reply.started":"2026-02-23T15:37:09.623308Z","shell.execute_reply":"2026-02-23T15:37:10.167871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Category imbalance score\n\nimbalance_ratio = train_class_df[\"count\"].max() / train_class_df[\"count\"].min()\nprint(\"Class Imbalance Ratio: \", imbalance_ratio)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T15:39:03.866471Z","iopub.execute_input":"2026-02-23T15:39:03.866894Z","iopub.status.idle":"2026-02-23T15:39:03.872872Z","shell.execute_reply.started":"2026-02-23T15:39:03.866863Z","shell.execute_reply":"2026-02-23T15:39:03.871993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport shutil\nfrom tqdm import tqdm","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q ultralytics timm","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE = \"/kaggle/input/vista26/Vistas Dataset Public/Vistas Dataset Public\"\nTRAIN_JSON = os.path.join(BASE, \"instances_train.json\")\nTRAIN_IMG_DIR = os.path.join(BASE, \"train\")\n\nYOLO_DIR = \"/kaggle/working/yolo_data\"\nos.makedirs(f\"{YOLO_DIR}/images/train\", exist_ok=True)\nos.makedirs(f\"{YOLO_DIR}/labels/train\", exist_ok=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(TRAIN_JSON) as f:\n    train_data = json.load(f)\n\nimages_dict = {img[\"id\"]: img for img in train_data[\"images\"]}\n\nfor ann in tqdm(train_data[\"annotations\"]):\n    img_info = images_dict[ann[\"image_id\"]]\n    img_name = img_info[\"file_name\"]\n\n    src = os.path.join(TRAIN_IMG_DIR, img_name)\n    dst = os.path.join(YOLO_DIR, \"images/train\", img_name)\n\n    if not os.path.exists(dst):\n        shutil.copy(src, dst)\n\n    w_img, h_img = img_info[\"width\"], img_info[\"height\"]\n    x, y, w, h = ann[\"bbox\"]\n\n    x_center = (x + w/2) / w_img\n    y_center = (y + h/2) / h_img\n    w_norm = w / w_img\n    h_norm = h / h_img\n\n    class_id = ann[\"category_id\"] - 1\n\n    label_path = os.path.join(YOLO_DIR, \"labels/train\", img_name.replace(\".jpg\", \".txt\"))\n    \n    with open(label_path, \"a\") as f:\n        f.write(f\"{class_id} {x_center} {y_center} {w_norm} {h_norm}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T18:04:28.877111Z","iopub.execute_input":"2026-02-23T18:04:28.877463Z","iopub.status.idle":"2026-02-23T18:04:28.895701Z","shell.execute_reply.started":"2026-02-23T18:04:28.877423Z","shell.execute_reply":"2026-02-23T18:04:28.894412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Creating YOLO config\n\ndata_yaml = f\"\"\"\npath: {YOLO_DIR}\ntrain: images/train\nnc: 200\nnames: {[str(i) for i in range(200)]}\n\"\"\"\n\nwith open(\"/kaggle/working/data.yaml\", \"w\") as f:\n    f.write(data_yaml)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Train YOLOv8\n\nfrom ultralytics import YOLO\n\nmodel = YOLO(\"yolov8m.pt\")\n\nmodel.train(\n    data = \"/kaggle/working/data.yaml\",\n    epochs = 30,\n    imgsz = 1024,\n    batch = 4,\n    device = 0,\n    amp = True,\n    mosaic = 1.0,\n    copy_paste = 0.5,\n    degrees = 10,\n    scale = 0.5,\n    fliplr = 0.5,\n    workers = 4\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train ResNet Classifier\n\nimport timm\nimport torch\nimport torch.nn as nn\nfrom torchvision import transforms\nfrom torch.utils.data import Dataset, DataLoader\nfrom PIL import Image\n\nclass ProductDataset(Dataset):\n    def __init__(self, images, labels):\n        self.images = images\n        self.labels = labels\n        self.transform = transforms.Compose([\n            transforms.Resize((224, 224)),\n            transforms.ToTensor(),\n        ])\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        img = Image.open(self.images[idx]).convert(\"RGB\")\n        img = self.transform(img)\n        label = self.labels[idx]\n        return img, label","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Model\n\nclassifier = timm.create_model(\n    \"resnet50\",\n    pretrained = True,\n    num_classes = 200\n)\nclassifier = classifier.cuda()\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(classifier.parameters(), lr=1e-4)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}