{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":85240,"databundleVersionId":9622164,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Setup and Preparation\n\n!pip install torch torchvision matplotlib seaborn opencv-python\n!git clone https://github.com/ultralytics/yolov5\n\n# Change to YOLOv5 directory\n%cd yolov5\n\n# Install YOLOv5 dependencies\n!pip install -r requirements.txt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T19:45:30.830241Z","iopub.execute_input":"2024-12-05T19:45:30.830514Z","iopub.status.idle":"2024-12-05T19:45:54.109297Z","shell.execute_reply.started":"2024-12-05T19:45:30.830487Z","shell.execute_reply":"2024-12-05T19:45:54.108264Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Data Preparation\n# Split the dataset into training and validation, ensuring balance among classes.\nimport os\nimport random\nimport shutil\nfrom pathlib import Path\nimport yaml\n\n# Define paths\nDATA_DIR = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train\"\nTEST_DIR = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test\"\nTRAIN_SPLIT = \"dataset/train_split\"\nVAL_SPLIT = \"dataset/val_split\"\n\n# Create directories\nos.makedirs(f\"{TRAIN_SPLIT}/images\", exist_ok=True)\nos.makedirs(f\"{TRAIN_SPLIT}/labels\", exist_ok=True)\nos.makedirs(f\"{VAL_SPLIT}/images\", exist_ok=True)\nos.makedirs(f\"{VAL_SPLIT}/labels\", exist_ok=True)\n\n# Get all images and corresponding label files\nimage_files = list(Path(f\"{DATA_DIR}/images\").glob(\"*.jpeg\"))\nlabel_files = list(Path(f\"{DATA_DIR}/labels\").glob(\"*.txt\"))\n\n# Sort to ensure matching\nimage_files.sort()\nlabel_files.sort()\n\n# Split into train(80%) and validation(20%)\nsplit_idx = int(0.8 * len(image_files))\nrandom.seed(42)\n\ntrain_images, val_images = image_files[:split_idx], image_files[split_idx:]\ntrain_labels, val_labels = label_files[:split_idx], label_files[split_idx:]\n\n# Move files to respective directories\nfor img, lbl in zip(train_images, train_labels):\n    shutil.copy(img, f\"{TRAIN_SPLIT}/images/{img.name}\")\n    shutil.copy(lbl, f\"{TRAIN_SPLIT}/labels/{lbl.name}\")\n\nfor img, lbl in zip(val_images, val_labels):\n    shutil.copy(img, f\"{VAL_SPLIT}/images/{img.name}\")\n    shutil.copy(lbl, f\"{VAL_SPLIT}/labels/{lbl.name}\")\n\nprint(f\"Training images: {len(train_images)}, Validation images: {len(val_images)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T19:46:03.439632Z","iopub.execute_input":"2024-12-05T19:46:03.440390Z","iopub.status.idle":"2024-12-05T19:49:54.848513Z","shell.execute_reply.started":"2024-12-05T19:46:03.440346Z","shell.execute_reply":"2024-12-05T19:49:54.847509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3: Create Dataset Configuration File\n# Create the dataset.yaml file to define the dataset paths and class labels.\nDATA_YAML = \"\"\"\ntrain: dataset/train_split/images\nval: dataset/val_split/images\n\nnc: 6\nnames: [\"aegypti\", \"albopictus\", \"anopheles\", \"culex\", \"culiseta\", \"japonicus/koreicus\"]\n\"\"\"\n\nwith open(\"dataset.yaml\", \"w\") as f:\n    f.write(DATA_YAML)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T19:49:54.850130Z","iopub.execute_input":"2024-12-05T19:49:54.850400Z","iopub.status.idle":"2024-12-05T19:49:54.855217Z","shell.execute_reply.started":"2024-12-05T19:49:54.850368Z","shell.execute_reply":"2024-12-05T19:49:54.854360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4: Modify Hyperparameters (Optional)\n# You can create a custom hyp.yaml file to adjust augmentation and training parameters.\nHYP_YAML = \"\"\"\nlr0: 0.001  # initial learning rate\nlrf: 0.2  # final learning rate (lr0 * lrf)\nmomentum: 0.937  # SGD momentum\nweight_decay: 0.0005  # optimizer weight decay\nwarmup_epochs: 3.0  # warmup epochs\nhsv_h: 0.015  # image HSV-Hue augmentation\nhsv_s: 0.7  # image HSV-Saturation augmentation\nhsv_v: 0.4  # image HSV-Value augmentation\ndegrees: 0.5  # image rotation (+/- deg)\ntranslate: 0.1  # image translation (+/- fraction)\nscale: 0.5  # image scale (+/- gain)\nshear: 0.1  # image shear (+/- deg)\nmosaic: 1.0  # mosaic augmentation probability\nmixup: 0.2  # mixup augmentation probability\n\"\"\"\n\nwith open(\"hyp.custom.yaml\", \"w\") as f:\n    f.write(HYP_YAML)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T19:49:54.856321Z","iopub.execute_input":"2024-12-05T19:49:54.856629Z","iopub.status.idle":"2024-12-05T19:49:54.874469Z","shell.execute_reply.started":"2024-12-05T19:49:54.856588Z","shell.execute_reply":"2024-12-05T19:49:54.873658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# To modify the built-in YAML file:\n\n# Navigate to yolov5/data/.\n# Copy hyp.scratch-low.yaml or hyp.scratch-high.yaml and modify it as above.","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%env WANDB_MODE = disabled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T19:55:23.919728Z","iopub.execute_input":"2024-12-05T19:55:23.920864Z","iopub.status.idle":"2024-12-05T19:55:23.926415Z","shell.execute_reply.started":"2024-12-05T19:55:23.920822Z","shell.execute_reply":"2024-12-05T19:55:23.925566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5: Train the Model\n# Train the YOLOv5l model using the custom hyp.yaml and dataset.yaml.\nimport torch\n\n# Check for GPU\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# Train the model\n!python train.py \\\n    --data dataset.yaml \\\n    --weights yolov5m.pt \\\n    --epochs 50 \\\n    --batch-size 32 \\\n    --img-size 640 \\\n    --project mosquito-detection \\\n    --name experiment \\\n    --device 0 \\\n    # --lr-scheduler cosine \\\n    --hyp hyp.custom.yaml\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T20:00:50.588885Z","iopub.execute_input":"2024-12-05T20:00:50.589796Z","iopub.status.idle":"2024-12-06T01:44:41.068221Z","shell.execute_reply.started":"2024-12-05T20:00:50.589758Z","shell.execute_reply":"2024-12-06T01:44:41.067252Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"training completed\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T01:44:41.070124Z","iopub.execute_input":"2024-12-06T01:44:41.070422Z","iopub.status.idle":"2024-12-06T01:44:41.075777Z","shell.execute_reply.started":"2024-12-06T01:44:41.070387Z","shell.execute_reply":"2024-12-06T01:44:41.074934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/working/yolov5/mosquito-detection/experiment2/weights","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:05:25.548638Z","iopub.execute_input":"2024-12-06T02:05:25.549068Z","iopub.status.idle":"2024-12-06T02:05:26.558463Z","shell.execute_reply.started":"2024-12-06T02:05:25.549018Z","shell.execute_reply":"2024-12-06T02:05:26.557436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6: Run Inference\n# Perform inference on the test dataset.\n!python detect.py \\\n    --weights mosquito-detection/experiment2/weights/best.pt \\\n    --source /kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images \\\n    --img-size 640 \\\n    --conf-thres 0.3 \\\n    --iou-thres 0.5 \\\n    --save-txt \\\n    --save-conf \\\n    --project mosquito-detection-results \\\n    --name inference \\\n    --agnostic-nms\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:05:44.324117Z","iopub.execute_input":"2024-12-06T02:05:44.324480Z","iopub.status.idle":"2024-12-06T02:06:52.787222Z","shell.execute_reply.started":"2024-12-06T02:05:44.324449Z","shell.execute_reply":"2024-12-06T02:06:52.786341Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate Submission File\nimport pandas as pd\nimport glob\nfrom collections import defaultdict\n\n# Path to YOLOv5 predictions\n\npredictions_dir = \"/kaggle/working/yolov5/mosquito-detection-results/inference2/labels\"\n# predictions_dir = \"/kaggle/working/mosquito-detection-results/inference/labels\"\n\nsubmission = []\nfor pred_file in glob.glob(os.path.join(predictions_dir, \"*.txt\")):\n    image_id = os.path.basename(pred_file).replace(\".txt\", \".jpeg\")\n    with open(pred_file, \"r\") as f:\n        for line in f:\n            label, x_center, y_center, width, height, conf = map(float, line.split())\n            submission.append({\n                \"id\": len(submission),\n                \"ImageID\": image_id,\n                \"LabelName\": [\"aegypti\", \"albopictus\", \"anopheles\", \"culex\", \"culiseta\", \"japonicus/koreicus\"][int(label)],\n                \"Conf\": conf,\n                \"xcenter\": x_center,\n                \"ycenter\": y_center,\n                \"bbx_width\": width,\n                \"bbx_height\": height\n            })\n\n\n\n# Save to CSV\nsubmission_df = pd.DataFrame(submission)\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:25:58.202417Z","iopub.execute_input":"2024-12-06T02:25:58.203420Z","iopub.status.idle":"2024-12-06T02:25:58.236269Z","shell.execute_reply.started":"2024-12-06T02:25:58.203377Z","shell.execute_reply":"2024-12-06T02:25:58.235548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_duplicate_count = 0\nfrom collections import Counter\n\n# Count the number of occurrences of each ImageID\nimage_counts = Counter(submission_df['ImageID'])\n\n# Print images with more than one prediction\nfor image_id, count in image_counts.items():\n    if count > 1:\n        print(f\"{image_id}: {count} predictions\")\n        total_duplicate_count += 1\n\nprint(total_duplicate_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:26:12.872838Z","iopub.execute_input":"2024-12-06T02:26:12.873206Z","iopub.status.idle":"2024-12-06T02:26:12.879085Z","shell.execute_reply.started":"2024-12-06T02:26:12.873180Z","shell.execute_reply":"2024-12-06T02:26:12.878182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter predictions to keep the highest confidence per class per image\nfiltered_submission = []\npredictions_by_image = defaultdict(list)\n\nfor row in submission:\n    predictions_by_image[row[\"ImageID\"]].append(row)\n\nfor image_id, preds in predictions_by_image.items():\n    unique_classes = defaultdict(list)\n    for pred in preds:\n        unique_classes[pred[\"LabelName\"]].append(pred)\n    for class_preds in unique_classes.values():\n        filtered_submission.append(max(class_preds, key=lambda x: x[\"Conf\"]))\n\n# Save to CSV\nsubmission_df = pd.DataFrame(filtered_submission)\n\nprint(f\"Number of predictions after filtering: {len(filtered_submission)}\")\n\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:27:49.107742Z","iopub.execute_input":"2024-12-06T02:27:49.108486Z","iopub.status.idle":"2024-12-06T02:27:49.124746Z","shell.execute_reply.started":"2024-12-06T02:27:49.108448Z","shell.execute_reply":"2024-12-06T02:27:49.124030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_duplicate_count = 0\nfrom collections import Counter\n\n# Count the number of occurrences of each ImageID\nimage_counts = Counter(submission_df['ImageID'])\n\n# Print images with more than one prediction\nfor image_id, count in image_counts.items():\n    if count > 1:\n        print(f\"{image_id}: {count} predictions\")\n        total_duplicate_count += 1\n\nprint(total_duplicate_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:28:07.166336Z","iopub.execute_input":"2024-12-06T02:28:07.166991Z","iopub.status.idle":"2024-12-06T02:28:07.172510Z","shell.execute_reply.started":"2024-12-06T02:28:07.166941Z","shell.execute_reply":"2024-12-06T02:28:07.171623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_image_ids = submission_df[\"ImageID\"].nunique()\ntotal_rows = len(submission_df)\nprint(f\"Unique ImageIDs: {unique_image_ids}\")\nprint(f\"Total rows: {total_rows}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:28:21.754282Z","iopub.execute_input":"2024-12-06T02:28:21.755049Z","iopub.status.idle":"2024-12-06T02:28:21.760858Z","shell.execute_reply.started":"2024-12-06T02:28:21.755013Z","shell.execute_reply":"2024-12-06T02:28:21.759988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pprint import pprint\n\n# Get all test image IDs\ntest_image_ids = set(os.listdir(\"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images\"))\ntest_image_ids = {img.replace(\".jpeg\", \"\") for img in test_image_ids}\n\n# Get predicted ImageIDs\npredicted_image_ids = set(submission_df[\"ImageID\"].str.replace(\".jpeg\", \"\"))\n\n# Find missing ImageIDs\nmissing_image_ids = test_image_ids - predicted_image_ids\nprint(f\"Missing ImageIDs: {len(missing_image_ids)}\")\npprint(missing_image_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:28:35.137302Z","iopub.execute_input":"2024-12-06T02:28:35.137651Z","iopub.status.idle":"2024-12-06T02:28:35.156199Z","shell.execute_reply.started":"2024-12-06T02:28:35.137621Z","shell.execute_reply":"2024-12-06T02:28:35.155358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction_files = set(\n    os.path.basename(f).replace(\".txt\", \"\") for f in glob.glob(os.path.join(predictions_dir, \"*.txt\"))\n)\nmissing_in_predictions = missing_image_ids - prediction_files\npprint(f\"Images without predictions: {missing_in_predictions}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:28:52.416549Z","iopub.execute_input":"2024-12-06T02:28:52.417245Z","iopub.status.idle":"2024-12-06T02:28:52.423955Z","shell.execute_reply.started":"2024-12-06T02:28:52.417210Z","shell.execute_reply":"2024-12-06T02:28:52.423187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\n\nduplicate_counts = Counter(submission_df[\"ImageID\"])\nduplicates = {img_id: count for img_id, count in duplicate_counts.items() if count > 1}\nprint(f\"Overpredicted ImageIDs: {len(duplicates)}\")\nfor img_id, count in duplicates.items():\n    print(f\"{img_id}: {count} predictions\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:29:11.557948Z","iopub.execute_input":"2024-12-06T02:29:11.558654Z","iopub.status.idle":"2024-12-06T02:29:11.563931Z","shell.execute_reply.started":"2024-12-06T02:29:11.558617Z","shell.execute_reply":"2024-12-06T02:29:11.563044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for img_id in duplicates.keys():\n    print(submission_df[submission_df[\"ImageID\"] == img_id])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:29:23.075201Z","iopub.execute_input":"2024-12-06T02:29:23.075822Z","iopub.status.idle":"2024-12-06T02:29:23.084179Z","shell.execute_reply.started":"2024-12-06T02:29:23.075789Z","shell.execute_reply":"2024-12-06T02:29:23.083251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\n# List of available class labels\navailable_classes = [\"aegypti\", \"albopictus\", \"anopheles\", \"culex\", \"culiseta\", \"japonicus/koreicus\"]\n\n\n# Add default predictions for missing ImageIDs\nfor img_id in missing_image_ids:\n    random_label = random.choice(available_classes)\n    submission.append({\n        \"id\": len(submission),\n        \"ImageID\": f\"{img_id}.jpeg\",\n        \"LabelName\": random_label,  # Assign a dummy label\n        \"Conf\": 0.0,  # Confidence of 0\n        \"xcenter\": 0.5,\n        \"ycenter\": 0.5,\n        \"bbx_width\": 0.1,\n        \"bbx_height\": 0.1\n    })\n\n# Create DataFrame\nsubmission_df = pd.DataFrame(submission)\n\nprint(f\"Final Unique ImageIDs: {submission_df['ImageID'].nunique()}\")\nprint(f\"Final Total Rows: {len(submission_df)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:29:43.816999Z","iopub.execute_input":"2024-12-06T02:29:43.817705Z","iopub.status.idle":"2024-12-06T02:29:43.825725Z","shell.execute_reply.started":"2024-12-06T02:29:43.817673Z","shell.execute_reply":"2024-12-06T02:29:43.824839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter predictions to keep the highest confidence prediction per image\nunique_submission = (\n    submission_df.loc[submission_df.groupby(\"ImageID\")[\"Conf\"].idxmax()]\n    .reset_index(drop=True)\n)\n\n\n\n# Print the final statistics\nprint(f\"Final Unique ImageIDs: {unique_submission['ImageID'].nunique()}\")\nprint(f\"Final Total Rows: {len(unique_submission)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:29:59.210971Z","iopub.execute_input":"2024-12-06T02:29:59.211328Z","iopub.status.idle":"2024-12-06T02:29:59.220327Z","shell.execute_reply.started":"2024-12-06T02:29:59.211299Z","shell.execute_reply":"2024-12-06T02:29:59.219616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:30:15.010977Z","iopub.execute_input":"2024-12-06T02:30:15.011656Z","iopub.status.idle":"2024-12-06T02:30:15.022529Z","shell.execute_reply.started":"2024-12-06T02:30:15.011618Z","shell.execute_reply":"2024-12-06T02:30:15.021819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Path to the sample submission file\nsample_submission_path = \"/kaggle/input/dlp-object-detection/sample_submission.csv\"\n\n# Read the CSV file\nsample_submission = pd.read_csv(sample_submission_path)\n\n# Display the first few rows to verify the content\nprint(sample_submission.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:39:33.126311Z","iopub.execute_input":"2024-12-06T02:39:33.126659Z","iopub.status.idle":"2024-12-06T02:39:33.144240Z","shell.execute_reply.started":"2024-12-06T02:39:33.126632Z","shell.execute_reply":"2024-12-06T02:39:33.143258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge submission with sample_submission to align ids\nfinal_submission = pd.merge(\n    unique_submission.drop(columns=[\"id\"], errors=\"ignore\"),  # Drop existing 'id' to prevent duplicates\n    sample_submission[[\"id\", \"ImageID\"]],\n    on=\"ImageID\",\n    how=\"left\"\n)\n\n# Reorder columns to place 'id' first\ncolumns_order = ['id'] + [col for col in final_submission.columns if col != 'id']\nfinal_submission = final_submission[columns_order]\nfinal_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:39:54.510139Z","iopub.execute_input":"2024-12-06T02:39:54.510924Z","iopub.status.idle":"2024-12-06T02:39:54.557044Z","shell.execute_reply.started":"2024-12-06T02:39:54.510876Z","shell.execute_reply":"2024-12-06T02:39:54.556286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the filtered submission to CSV\nunique_submission = final_submission.copy()\nunique_submission.to_csv(\"notebook6_submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:40:19.186405Z","iopub.execute_input":"2024-12-06T02:40:19.187428Z","iopub.status.idle":"2024-12-06T02:40:19.197012Z","shell.execute_reply.started":"2024-12-06T02:40:19.187383Z","shell.execute_reply":"2024-12-06T02:40:19.195959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:40:30.970485Z","iopub.execute_input":"2024-12-06T02:40:30.971191Z","iopub.status.idle":"2024-12-06T02:40:30.982212Z","shell.execute_reply.started":"2024-12-06T02:40:30.971156Z","shell.execute_reply":"2024-12-06T02:40:30.981391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(unique_submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T02:31:07.992461Z","iopub.execute_input":"2024-12-06T02:31:07.993125Z","iopub.status.idle":"2024-12-06T02:31:07.998766Z","shell.execute_reply.started":"2024-12-06T02:31:07.993090Z","shell.execute_reply":"2024-12-06T02:31:07.997923Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# END Of NOTEBOOK","metadata":{}}]}