{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":128792,"databundleVersionId":15494745,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:05:03.322973Z","iopub.execute_input":"2026-01-31T17:05:03.323298Z","iopub.status.idle":"2026-01-31T17:05:09.036595Z","shell.execute_reply.started":"2026-01-31T17:05:03.323265Z","shell.execute_reply":"2026-01-31T17:05:09.035879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, json, cv2, glob\nfrom pathlib import Path\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom ultralytics import YOLO","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:05:09.038546Z","iopub.execute_input":"2026-01-31T17:05:09.038854Z","iopub.status.idle":"2026-01-31T17:05:17.418254Z","shell.execute_reply.started":"2026-01-31T17:05:09.038825Z","shell.execute_reply":"2026-01-31T17:05:17.417691Z"},"collapsed":true,"jupyter":{"outputs_hidden":true,"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE = \"/kaggle/input/vista26/Vistas Dataset Public/Vistas Dataset Public\"\nTRAIN_JSON = f\"{BASE}/instances_train.json\"\nCAT_JSON = f\"{BASE}/Categories.json\"\nTRAIN_IMG_DIR = f\"{BASE}/train\"\n\nWORK = Path(\"/kaggle/working\")\nIMG_TRAIN, LBL_TRAIN = WORK/\"images/train\", WORK/\"labels/train\"\nIMG_VAL, LBL_VAL = WORK/\"images/val\", WORK/\"labels/val\"\nfor p in [IMG_TRAIN, LBL_TRAIN, IMG_VAL, LBL_VAL]: p.mkdir(parents=True, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:05:17.419188Z","iopub.execute_input":"2026-01-31T17:05:17.419731Z","iopub.status.idle":"2026-01-31T17:05:17.425684Z","shell.execute_reply.started":"2026-01-31T17:05:17.419692Z","shell.execute_reply":"2026-01-31T17:05:17.424940Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(CAT_JSON) as f:\n    cats = json.load(f)[\"categories\"]\nid_map = {c[\"id\"]: i for i, c in enumerate(cats)}\nnum_classes = len(cats)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:05:17.426763Z","iopub.execute_input":"2026-01-31T17:05:17.426996Z","iopub.status.idle":"2026-01-31T17:05:17.449762Z","shell.execute_reply.started":"2026-01-31T17:05:17.426974Z","shell.execute_reply":"2026-01-31T17:05:17.449107Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(TRAIN_JSON) as f:\n    data = json.load(f)\n\nimages = data[\"images\"]\ntrain_imgs, val_imgs = train_test_split(images, test_size=0.2, random_state=42)\ntrain_ids = {i[\"id\"] for i in train_imgs}\nval_ids = {i[\"id\"] for i in val_imgs}\n\ntrain_data = {\n    \"images\": train_imgs,\n    \"annotations\": [a for a in data[\"annotations\"] if a[\"image_id\"] in {img[\"id\"] for img in train_imgs}],\n    \"categories\": cats  # use Categories.json instead of data[\"categories\"]\n}\n\nval_data = {\n    \"images\": val_imgs,\n    \"annotations\": [a for a in data[\"annotations\"] if a[\"image_id\"] in {img[\"id\"] for img in val_imgs}],\n    \"categories\": cats  # use Categories.json instead of data[\"categories\"]\n}\n\nTRAIN_JSON_SPLIT, VAL_JSON_SPLIT = WORK/\"train.json\", WORK/\"val.json\"\nwith open(TRAIN_JSON_SPLIT, \"w\") as f: json.dump(train_data, f)\nwith open(VAL_JSON_SPLIT, \"w\") as f: json.dump(val_data, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:05:17.450628Z","iopub.execute_input":"2026-01-31T17:05:17.451201Z","iopub.status.idle":"2026-01-31T17:12:51.046022Z","shell.execute_reply.started":"2026-01-31T17:05:17.451176Z","shell.execute_reply":"2026-01-31T17:12:51.045377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_coco_to_yolo(json_file, img_dir, out_img_dir, out_lbl_dir):\n    with open(json_file) as f: data = json.load(f)\n    images_dict = {img[\"id\"]: img for img in data[\"images\"]}\n\n    for ann in data[\"annotations\"]:\n        img_id = ann[\"image_id\"]\n        bbox = ann[\"bbox\"]\n        cat_id = id_map[ann[\"category_id\"]]\n\n        img_info = images_dict[img_id]\n        img_path = f\"{img_dir}/{img_info['file_name']}\"\n        if not os.path.exists(img_path): continue\n\n        img = cv2.imread(img_path)\n        if img is None: continue\n\n        h, w = img.shape[:2]\n        x, y, bw, bh = bbox\n        xc, yc, bw, bh = (x+bw/2)/w, (y+bh/2)/h, bw/w, bh/h\n\n        lbl_path = out_lbl_dir / f\"{Path(img_info['file_name']).stem}.txt\"\n        with open(lbl_path, \"a\") as f:\n            f.write(f\"{cat_id} {xc} {yc} {bw} {bh}\\n\")\n\n        # Symlink instead of copy to save space\n        dst = out_img_dir / img_info[\"file_name\"]\n        if not dst.exists():\n            os.symlink(img_path, dst)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:12:51.047836Z","iopub.execute_input":"2026-01-31T17:12:51.048085Z","iopub.status.idle":"2026-01-31T17:12:51.054789Z","shell.execute_reply.started":"2026-01-31T17:12:51.048058Z","shell.execute_reply":"2026-01-31T17:12:51.054189Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"convert_coco_to_yolo(TRAIN_JSON_SPLIT, TRAIN_IMG_DIR, IMG_TRAIN, LBL_TRAIN)\nconvert_coco_to_yolo(VAL_JSON_SPLIT, TRAIN_IMG_DIR, IMG_VAL, LBL_VAL)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:12:51.055753Z","iopub.execute_input":"2026-01-31T17:12:51.056078Z","iopub.status.idle":"2026-01-31T17:33:50.261989Z","shell.execute_reply.started":"2026-01-31T17:12:51.056052Z","shell.execute_reply":"2026-01-31T17:33:50.261200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yaml_path = WORK / \"data.yaml\"\nwith open(yaml_path, \"w\") as f:\n    f.write(f\"path: /kaggle/working\\n\")\n    f.write(f\"train: images/train\\nval: images/val\\nnc: {num_classes}\\nnames:\\n\")\n    for i, cat in enumerate(cats): f.write(f\"  {i}: {cat['name']}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:33:50.263112Z","iopub.execute_input":"2026-01-31T17:33:50.263427Z","iopub.status.idle":"2026-01-31T17:33:50.268711Z","shell.execute_reply.started":"2026-01-31T17:33:50.263391Z","shell.execute_reply":"2026-01-31T17:33:50.268116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = YOLO(\"yolov8n.pt\")  \nmodel.train(\n    data=str(yaml_path),\n    epochs=12,\n    imgsz=512,\n    batch=14,\n    cache=False,\n    save=False,\n    plots=False,\n    workers=2\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-31T17:33:50.269746Z","iopub.execute_input":"2026-01-31T17:33:50.270164Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_images = sorted(glob.glob(f\"{IMG_VAL}/*.jpg\"))\nrows = []\n\nfor img_path in val_images:\n    img_name = Path(img_path).stem\n    try: image_id = int(img_name.split(\"-\")[-1])\n    except: image_id = img_name\n\n    results = model(img_path, verbose=False)[0]\n    classes = results.boxes.cls.cpu().numpy().astype(int).tolist()\n    classes.sort()\n    rows.append({\"image_id\": image_id, \"categories\": str(classes)})\n\ndf = pd.DataFrame(rows).sort_values(\"image_id\")\ndf.to_csv(\"submission.csv\", index=False)\n\nprint(\"✅ submission.csv created!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}