{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":86142,"databundleVersionId":9786425,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Sample Submission Notebook\nThis notebook uses the YOLO11n-seg model provided by Ultralytics to segment the test images. These images are then converted into the required submission format.\n\nNote: This is obviously a very simple pipeline just to showcase the submission format, the expected pipeline should be much better.","metadata":{}},{"cell_type":"markdown","source":"## Model and Predictions\nObviously this code is YOLO specific, the ultimate goal of this section is to produce the results, and the next section will focus on understanding the submission file and converting the results of the YOLO model to the required results format.","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics -q","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:12.576170Z","iopub.execute_input":"2024-10-08T09:33:12.576597Z","iopub.status.idle":"2024-10-08T09:33:30.142772Z","shell.execute_reply.started":"2024-10-08T09:33:12.576557Z","shell.execute_reply":"2024-10-08T09:33:30.141302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ultralytics import YOLO","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:30.145473Z","iopub.execute_input":"2024-10-08T09:33:30.145968Z","iopub.status.idle":"2024-10-08T09:33:34.625453Z","shell.execute_reply.started":"2024-10-08T09:33:30.145914Z","shell.execute_reply":"2024-10-08T09:33:34.624252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO(model=\"yolo11n-seg.pt\")","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:34.626776Z","iopub.execute_input":"2024-10-08T09:33:34.627305Z","iopub.status.idle":"2024-10-08T09:33:36.579150Z","shell.execute_reply.started":"2024-10-08T09:33:34.627262Z","shell.execute_reply":"2024-10-08T09:33:36.578169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model('/kaggle/input/iitg-ai-overnight-hackathon-2024/dataset/dataset/test')","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:36.582228Z","iopub.execute_input":"2024-10-08T09:33:36.582732Z","iopub.status.idle":"2024-10-08T09:33:59.400986Z","shell.execute_reply.started":"2024-10-08T09:33:36.582676Z","shell.execute_reply":"2024-10-08T09:33:59.400009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(results[0])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:59.402763Z","iopub.execute_input":"2024-10-08T09:33:59.403641Z","iopub.status.idle":"2024-10-08T09:33:59.410150Z","shell.execute_reply.started":"2024-10-08T09:33:59.403592Z","shell.execute_reply":"2024-10-08T09:33:59.409235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can go through each field of the output (results[0]) to understand what each field means in detail, we are interested in the .masks object.\n\nThe results[0].masks contain polygon masks of each detected image.\n\nFor the first image, you can run len(results[0].masks.xy) to get 16, which suggests there are 16 detected objects (with confidences given in results[0].boxes).\n\nWe need to use this results[0].masks.xy[0] and so on to get the required polygon to put into the submission.csv file. This is done below after a brief discussion on the submission format.","metadata":{}},{"cell_type":"code","source":"print('The polygon mask for the first object in the first image is ', results[0].masks.xy[0]) # Polygon mask of per Object per file","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:59.411893Z","iopub.execute_input":"2024-10-08T09:33:59.412563Z","iopub.status.idle":"2024-10-08T09:33:59.447094Z","shell.execute_reply.started":"2024-10-08T09:33:59.412518Z","shell.execute_reply":"2024-10-08T09:33:59.446245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The first object class id is', results[0].boxes.cls[0].item(), \n      'with confidence', results[0].boxes.conf[0].item(),\n      'and, the cooresponding object is a', results[0].names[int(results[0].boxes.cls[0].item())])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:59.448362Z","iopub.execute_input":"2024-10-08T09:33:59.449378Z","iopub.status.idle":"2024-10-08T09:33:59.457639Z","shell.execute_reply.started":"2024-10-08T09:33:59.449333Z","shell.execute_reply":"2024-10-08T09:33:59.456489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Converting to required submission format","metadata":{}},{"cell_type":"markdown","source":"### Understanding the submission format","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nsample = pd.read_csv('/kaggle/input/iitg-ai-overnight-hackathon-2024/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:59.459148Z","iopub.execute_input":"2024-10-08T09:33:59.459564Z","iopub.status.idle":"2024-10-08T09:33:59.872103Z","shell.execute_reply.started":"2024-10-08T09:33:59.459506Z","shell.execute_reply":"2024-10-08T09:33:59.871053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:59.873561Z","iopub.execute_input":"2024-10-08T09:33:59.873925Z","iopub.status.idle":"2024-10-08T09:33:59.896309Z","shell.execute_reply.started":"2024-10-08T09:33:59.873886Z","shell.execute_reply":"2024-10-08T09:33:59.895130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(sample['objects'][0])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:59.899983Z","iopub.execute_input":"2024-10-08T09:33:59.901193Z","iopub.status.idle":"2024-10-08T09:33:59.908438Z","shell.execute_reply.started":"2024-10-08T09:33:59.901145Z","shell.execute_reply":"2024-10-08T09:33:59.907269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\njson.loads(sample['objects'][0])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:59.909877Z","iopub.execute_input":"2024-10-08T09:33:59.910280Z","iopub.status.idle":"2024-10-08T09:33:59.930718Z","shell.execute_reply.started":"2024-10-08T09:33:59.910239Z","shell.execute_reply":"2024-10-08T09:33:59.929278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the sample submission file, each entry is indexed by the filename (except the part after _).\n\nThe objects field is a string of a list, each entry in the list consist of label and polygon representing the xy coordinates of the points ({'label': 'road', 'polygon': [[1224, 399], [585, 87], [1378, 852]]})","metadata":{}},{"cell_type":"markdown","source":"### Converting to the required format","metadata":{}},{"cell_type":"code","source":"submissionDict = {}","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:33:59.932275Z","iopub.execute_input":"2024-10-08T09:33:59.932701Z","iopub.status.idle":"2024-10-08T09:33:59.939901Z","shell.execute_reply.started":"2024-10-08T09:33:59.932657Z","shell.execute_reply":"2024-10-08T09:33:59.938739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for result in results:\n    filename = result.path.split('/')[-1].split('_')[0]   # Getting the filename i.e. id field\n    \n    # Populating data to put into the json using result\n    \n    #Handling no objects detected\n    if result.masks == None:\n        submissionDict[filename] = '[]' # Putting empty list\n        continue\n    \n    objectsDetected = '[' # Instead of creating a list and the converting it to string\n                          # directly create the string and append json strings to it\n\n    for i in range(len(result.masks.xy)): # Iterating through all detected objects\n        # Creating temporary dictionary for that object\n        objectSpecificData = {\n            'label': 'xyz',\n            'polygon': []\n        }\n        \n        # Populating objects data\n        objectSpecificData['label'] = result.names[int(result.boxes.cls[i].item())]\n        objectSpecificData['polygon'] = result.masks.xy[i].tolist()\n        \n        # Appending the json string of this dictionary to the string of this file\n        objectsDetected += json.dumps(objectSpecificData)\n        \n        if i != (len(result.masks.xy)-1):\n            objectsDetected += ','\n\n    # Closing list of file specific list\n    objectsDetected += ']'\n\n    # Adding the string of list of jsons to the submission dictionary\n    submissionDict[filename] = objectsDetected","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:36:16.083331Z","iopub.execute_input":"2024-10-08T09:36:16.083788Z","iopub.status.idle":"2024-10-08T09:36:16.329128Z","shell.execute_reply.started":"2024-10-08T09:36:16.083745Z","shell.execute_reply":"2024-10-08T09:36:16.327826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final Result\nsubmissionDict['frame0000']","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:36:19.721667Z","iopub.execute_input":"2024-10-08T09:36:19.722136Z","iopub.status.idle":"2024-10-08T09:36:19.729781Z","shell.execute_reply.started":"2024-10-08T09:36:19.722092Z","shell.execute_reply":"2024-10-08T09:36:19.728406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating submission.csv\nsubmission = pd.DataFrame(list(submissionDict.items()), columns=['id', 'objects'])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:34:00.233852Z","iopub.execute_input":"2024-10-08T09:34:00.234268Z","iopub.status.idle":"2024-10-08T09:34:00.241403Z","shell.execute_reply.started":"2024-10-08T09:34:00.234218Z","shell.execute_reply":"2024-10-08T09:34:00.240252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:34:00.243173Z","iopub.execute_input":"2024-10-08T09:34:00.243583Z","iopub.status.idle":"2024-10-08T09:34:00.259501Z","shell.execute_reply.started":"2024-10-08T09:34:00.243542Z","shell.execute_reply":"2024-10-08T09:34:00.257924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T09:34:00.261244Z","iopub.execute_input":"2024-10-08T09:34:00.261638Z","iopub.status.idle":"2024-10-08T09:34:00.291724Z","shell.execute_reply.started":"2024-10-08T09:34:00.261599Z","shell.execute_reply":"2024-10-08T09:34:00.290145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# All The Best!","metadata":{}}]}