{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":128792,"databundleVersionId":15494745,"sourceType":"competition"},{"sourceId":738203,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":562948,"modelId":575510}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -U transformers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T04:15:23.594943Z","iopub.execute_input":"2026-02-02T04:15:23.595232Z","iopub.status.idle":"2026-02-02T04:15:39.594667Z","shell.execute_reply.started":"2026-02-02T04:15:23.595199Z","shell.execute_reply":"2026-02-02T04:15:39.593694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##kunal version","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from huggingface_hub import login\nlogin()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T04:15:39.596042Z","iopub.execute_input":"2026-02-02T04:15:39.59636Z","iopub.status.idle":"2026-02-02T04:15:40.175814Z","shell.execute_reply.started":"2026-02-02T04:15:39.59632Z","shell.execute_reply":"2026-02-02T04:15:40.175002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T04:17:02.601927Z","iopub.execute_input":"2026-02-02T04:17:02.602575Z","iopub.status.idle":"2026-02-02T04:17:02.606164Z","shell.execute_reply.started":"2026-02-02T04:17:02.602544Z","shell.execute_reply":"2026-02-02T04:17:02.605332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers.models.sam3.processing_sam3 import Sam3Processor\n\nprocessor = Sam3Processor.from_pretrained(\"facebook/sam3\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T04:17:06.442403Z","iopub.execute_input":"2026-02-02T04:17:06.443012Z","iopub.status.idle":"2026-02-02T04:17:09.483561Z","shell.execute_reply.started":"2026-02-02T04:17:06.44298Z","shell.execute_reply":"2026-02-02T04:17:09.482983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers.models.sam3.modeling_sam3 import Sam3Model\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = Sam3Model.from_pretrained(\"facebook/sam3\").to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T04:18:20.222821Z","iopub.execute_input":"2026-02-02T04:18:20.223569Z","iopub.status.idle":"2026-02-02T04:18:41.597654Z","shell.execute_reply.started":"2026-02-02T04:18:20.223537Z","shell.execute_reply":"2026-02-02T04:18:41.596992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========================================\n# PREDICT LABELS ON SEGMENTED MASKS\n# ========================================\n\nimport os\nfrom pathlib import Path\nfrom PIL import Image\nimport torch\nfrom torchvision import transforms, models\nimport torch.nn as nn\n\n# Setup\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmask_folder = \"segmented_masks\"\npredictions_list = []\n\n# Use the same validation transforms as training\nmean = [0.485, 0.456, 0.406]\nstd  = [0.229, 0.224, 0.225]\n\npredict_tfms = transforms.Compose([\n    transforms.Resize(256),\n    transforms.CenterCrop(224),\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std),\n])\n\n# ========================================\n# LOAD TRAINED CONVNEXT-BASE MODEL\n# ========================================\n\nnum_classes = 200\n\n# Build the same architecture used in training\nconvnext_model = models.convnext_base(weights=None)   # no pretrained weights, we load our own\n\n# Replace the classifier head — MUST match training exactly\nnum_ftrs = convnext_model.classifier[-1].in_features  # 1024 for ConvNeXt-Base\nconvnext_model.classifier[-1] = nn.Sequential(\n    nn.Dropout(0.4),\n    nn.Linear(num_ftrs, 512),\n    nn.GELU(),\n    nn.Dropout(0.2),\n    nn.Linear(512, num_classes)\n)\n\n# Load the trained weights\nmodel_path = '/kaggle/input/convexnet-retail/pytorch/default/1/best_model.pth'\nconvnext_model.load_state_dict(torch.load(model_path, map_location=device, weights_only=True))\nconvnext_model = convnext_model.to(device)\nconvnext_model.eval()\n\nprint(f\"✓ Loaded trained ConvNeXt-Base model from {model_path}\")\n\nclass_names= ['1','10','100','101','102','103','104', '105','106','107','108', '109', '11',\n '110','111','112','113',\n '114',\n '115',\n '116',\n '117',\n '118',\n '119',\n '12',\n '120',\n '121',\n '122',\n '123',\n '124',\n '125',\n '126',\n '127',\n '128',\n '129',\n '13',\n '130',\n '131',\n '132',\n '133',\n '134',\n '135',\n '136',\n '137',\n '138',\n '139',\n '14',\n '140',\n '141',\n '142',\n '143',\n '144',\n '145',\n '146',\n '147',\n '148',\n '149',\n '15',\n '150',\n '151',\n '152',\n '153',\n '154',\n '155',\n '156',\n '157',\n '158',\n '159',\n '16',\n '160',\n '161',\n '162',\n '163',\n '164',\n '165',\n '166',\n '167',\n '168',\n '169',\n '17',\n '170',\n '171',\n '172',\n '173',\n '174',\n '175',\n '176',\n '177',\n '178',\n '179',\n '18',\n '180',\n '181',\n '182',\n '183',\n '184',\n '185',\n '186',\n '187',\n '188',\n '189',\n '19',\n '190',\n '191',\n '192',\n '193',\n '194',\n '195',\n '196',\n '197',\n '198',\n '199',\n '2',\n '20',\n '200',\n '21',\n '22',\n '23',\n '24',\n '25',\n '26',\n '27',\n '28',\n '29',\n '3',\n '30',\n '31',\n '32',\n '33',\n '34',\n '35',\n '36',\n '37',\n '38',\n '39',\n '4',\n '40',\n '41',\n '42',\n '43',\n '44',\n '45',\n '46',\n '47',\n '48',\n '49',\n '5',\n '50',\n '51',\n '52',\n '53',\n '54',\n '55',\n '56',\n '57',\n '58',\n '59',\n '6',\n '60',\n '61',\n '62',\n '63',\n '64',\n '65',\n '66',\n '67',\n '68',\n '69',\n '7',\n '70',\n '71',\n '72',\n '73',\n '74',\n '75',\n '76',\n '77',\n '78',\n '79',\n '8',\n '80',\n '81',\n '82',\n '83',\n '84',\n '85',\n '86',\n '87',\n '88',\n '89',\n '9',\n '90',\n '91',\n '92',\n '93',\n '94',\n '95',\n '96',\n '97',\n '98',\n '99']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T04:18:41.599093Z","iopub.execute_input":"2026-02-02T04:18:41.599408Z","iopub.status.idle":"2026-02-02T04:18:47.112503Z","shell.execute_reply.started":"2026-02-02T04:18:41.599382Z","shell.execute_reply":"2026-02-02T04:18:47.111876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def update_json_dict(json_path, key, value_list):\n    # Load existing dictionary\n    if os.path.exists(json_path) and os.path.getsize(json_path) > 0:\n        with open(json_path, \"r\") as f:\n            data = json.load(f)\n    else:\n        data = {}\n\n    # Replace or insert\n    data[key] = value_list\n\n    # Write back\n    with open(json_path, \"w\") as f:\n        json.dump(data, f, indent=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T04:18:47.113325Z","iopub.execute_input":"2026-02-02T04:18:47.113603Z","iopub.status.idle":"2026-02-02T04:18:54.375637Z","shell.execute_reply.started":"2026-02-02T04:18:47.113578Z","shell.execute_reply":"2026-02-02T04:18:54.374794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport json\nfrom PIL import Image\n\ntest_images_folder = \"/kaggle/input/vista26/Vistas Dataset Public/Vistas Dataset Public/validation\"\nstart_idx = 0      # change\nend_idx =  6001     # change (exclusive)\n\n\n\n# get sorted image list\nimage_files = sorted([\n    f for f in os.listdir(test_images_folder)\n    if f.lower().endswith(('.jpg', '.png', '.jpeg'))\n])\n\n# loop over selected range\nfor image_file in image_files[start_idx:end_idx]:\n\n    image_path = os.path.join(test_images_folder, image_file)\n    image = Image.open(image_path).convert(\"RGB\")\n\n    base_image_file_name = os.path.basename(image_path)\n\n    # segment using text prompt\n    inputs = processor(\n        images=image,\n        text=\"packaged retail food item\",\n        return_tensors=\"pt\"\n    ).to(device)\n\n    with torch.no_grad():\n        outputs = model(**inputs)\n\n    results = processor.post_process_instance_segmentation(\n        outputs,\n        threshold=0.3,\n        mask_threshold=0.3,\n        target_sizes=inputs.get(\"original_sizes\").tolist()\n    )[0]\n\n    print(f\"{base_image_file_name} → Found {len(results['masks'])} objects\")\n\n    predictions_list = []\n\n    # Original image once\n    img_np = np.array(image)\n\n    for idx, mask in enumerate(results['masks']):\n\n        # mask → numpy\n        mask_np = mask.cpu().numpy() if isinstance(mask, torch.Tensor) else mask\n\n        # white background\n        h, w = mask_np.shape\n        white_bg = np.ones((h, w, 3), dtype=np.uint8) * 255\n\n        # binary mask\n        binary_mask = mask_np > 0.5\n\n        # apply mask\n        masked_img = white_bg.copy()\n        masked_img[binary_mask] = img_np[binary_mask]\n\n        # PIL → tensor\n        masked_pil = Image.fromarray(masked_img)\n        img_tensor = predict_tfms(masked_pil).unsqueeze(0).to(device)\n\n        # classify\n        with torch.no_grad():\n            outputs = convnext_model(img_tensor)\n            probabilities = torch.nn.functional.softmax(outputs, dim=1)\n            confidence, predicted_idx = torch.max(probabilities, 1)\n\n        predictions_list.append({\n            'predicted_class': class_names[predicted_idx.item()],\n        })\n\n    predicted_classes = sorted(\n        int(p[\"predicted_class\"]) for p in predictions_list\n    )\n    update_json_dict('output.json',base_image_file_name,predicted_classes)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-02T04:19:45.930168Z","iopub.execute_input":"2026-02-02T04:19:45.930806Z"}},"outputs":[],"execution_count":null}]}