{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"###Importing Libraries","metadata":{"id":"rDH3mL6msKRZ"}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\nimport glob\nimport shutil","metadata":{"papermill":{"duration":0.191779,"end_time":"2023-02-04T15:44:15.014295","exception":false,"start_time":"2023-02-04T15:44:14.822516","status":"completed"},"tags":[],"id":"d0XY6eUhD5AC","execution":{"iopub.status.busy":"2023-06-19T20:41:18.112246Z","iopub.execute_input":"2023-06-19T20:41:18.11322Z","iopub.status.idle":"2023-06-19T20:41:18.300555Z","shell.execute_reply.started":"2023-06-19T20:41:18.113174Z","shell.execute_reply":"2023-06-19T20:41:18.299188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###Connecting to Google Drive","metadata":{"papermill":{"duration":0.010122,"end_time":"2023-02-04T15:44:15.035163","exception":false,"start_time":"2023-02-04T15:44:15.025041","status":"completed"},"tags":[],"id":"tK1bN12PD5AE"}},{"cell_type":"code","source":"# from google.colab import drive\n# drive.mount('/content/drive')","metadata":{"id":"NlboCJVUE5If","outputId":"0ca473e9-3e15-499a-d247-80ac05d9a7c3"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading the Data","metadata":{"papermill":{"duration":0.01037,"end_time":"2023-02-04T15:44:15.057314","exception":false,"start_time":"2023-02-04T15:44:15.046944","status":"completed"},"tags":[],"id":"mhQn0fbND5AF"}},{"cell_type":"code","source":"#reading the segmentations of the train data\ntrain_segmentation = pd.read_csv(\"/kaggle/input/airbus-ship-detection/train_ship_segmentations_v2.csv\")\n#changing the EncodedPixels column to string\ntrain_segmentation['EncodedPixels'] = train_segmentation['EncodedPixels'].astype(\"string\")","metadata":{"papermill":{"duration":1.285796,"end_time":"2023-02-04T15:44:16.35342","exception":false,"start_time":"2023-02-04T15:44:15.067624","status":"completed"},"tags":[],"id":"ah3ypz4ND5AF","execution":{"iopub.status.busy":"2023-06-19T20:41:35.338046Z","iopub.execute_input":"2023-06-19T20:41:35.338396Z","iopub.status.idle":"2023-06-19T20:41:36.383615Z","shell.execute_reply.started":"2023-06-19T20:41:35.338371Z","shell.execute_reply":"2023-06-19T20:41:36.38262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Corrupted images","metadata":{"papermill":{"duration":0.011642,"end_time":"2023-02-04T15:50:47.577235","exception":false,"start_time":"2023-02-04T15:50:47.565593","status":"completed"},"tags":[],"id":"jB6VaL2cD5AH"}},{"cell_type":"code","source":"#check the information for the corrupted images specifically.\ncorrupted_images = ['6384c3e78.jpg']\ncorrupted_rows = train_segmentation[train_segmentation['ImageId'].isin(corrupted_images)]\ncorrupted_rows","metadata":{"papermill":{"duration":0.043475,"end_time":"2023-02-04T15:50:47.651562","exception":false,"start_time":"2023-02-04T15:50:47.608087","status":"completed"},"tags":[],"id":"G-bCRk4ID5AH","outputId":"2621c956-2484-4d41-abe8-6fd2533cd614","execution":{"iopub.status.busy":"2023-06-19T20:41:40.577429Z","iopub.execute_input":"2023-06-19T20:41:40.57779Z","iopub.status.idle":"2023-06-19T20:41:40.618729Z","shell.execute_reply.started":"2023-06-19T20:41:40.577762Z","shell.execute_reply":"2023-06-19T20:41:40.617663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#deleting the corrupted image.\ntrain_segmentation = train_segmentation.drop(corrupted_rows.index)\ncorrupted_rows","metadata":{"papermill":{"duration":0.05404,"end_time":"2023-02-04T15:50:47.737307","exception":false,"start_time":"2023-02-04T15:50:47.683267","status":"completed"},"tags":[],"id":"5EvdXt0XD5AH","outputId":"8bf8361c-5440-4e55-898c-b659a7a01cf7","execution":{"iopub.status.busy":"2023-06-19T20:41:43.839292Z","iopub.execute_input":"2023-06-19T20:41:43.839646Z","iopub.status.idle":"2023-06-19T20:41:43.889574Z","shell.execute_reply.started":"2023-06-19T20:41:43.839619Z","shell.execute_reply":"2023-06-19T20:41:43.888513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Exploration","metadata":{"papermill":{"duration":0.010586,"end_time":"2023-02-04T15:50:47.758924","exception":false,"start_time":"2023-02-04T15:50:47.748338","status":"completed"},"tags":[],"id":"ikwtda3LD5AI"}},{"cell_type":"code","source":"print(f'Number of rows in the data - {train_segmentation.shape[0]}')","metadata":{"papermill":{"duration":0.026796,"end_time":"2023-02-04T15:50:47.796223","exception":false,"start_time":"2023-02-04T15:50:47.769427","status":"completed"},"tags":[],"id":"FMsHJXp3D5AI","outputId":"5f979bc0-1478-4c2f-bdba-59c657e95446","execution":{"iopub.status.busy":"2023-06-19T20:41:52.21964Z","iopub.execute_input":"2023-06-19T20:41:52.220032Z","iopub.status.idle":"2023-06-19T20:41:52.225809Z","shell.execute_reply.started":"2023-06-19T20:41:52.220001Z","shell.execute_reply":"2023-06-19T20:41:52.224969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preview of the data in train_segmentation\ntrain_segmentation.head(20)","metadata":{"id":"Ij7_Q_aq2X1E","outputId":"036503dc-bffa-4b40-c3d3-10d34df67ab6","execution":{"iopub.status.busy":"2023-06-19T20:41:55.268805Z","iopub.execute_input":"2023-06-19T20:41:55.269173Z","iopub.status.idle":"2023-06-19T20:41:55.280431Z","shell.execute_reply.started":"2023-06-19T20:41:55.269126Z","shell.execute_reply":"2023-06-19T20:41:55.279162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculating the number of unique images in the 'ImageId' column of the 'train_segmentation'\ntrain_images_number = train_segmentation['ImageId'].nunique()\nprint(f'Number of Train Images - {train_images_number}')","metadata":{"papermill":{"duration":0.055467,"end_time":"2023-02-04T15:50:47.862679","exception":false,"start_time":"2023-02-04T15:50:47.807212","status":"completed"},"tags":[],"id":"r7W6wJBXD5AI","outputId":"f9afcd90-7e40-426f-d43f-78ecfe8c80b5","execution":{"iopub.status.busy":"2023-06-19T20:42:00.009822Z","iopub.execute_input":"2023-06-19T20:42:00.010216Z","iopub.status.idle":"2023-06-19T20:42:00.068606Z","shell.execute_reply.started":"2023-06-19T20:42:00.01018Z","shell.execute_reply":"2023-06-19T20:42:00.067508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculate the number of images in 'train_segmentation' that do not have ship annotations\nimages_without_ships = train_segmentation['EncodedPixels'].isna().sum()\nprint(f'Number of images without ships - {images_without_ships}')","metadata":{"papermill":{"duration":0.03192,"end_time":"2023-02-04T15:50:48.00531","exception":false,"start_time":"2023-02-04T15:50:47.97339","status":"completed"},"tags":[],"id":"LjqQGi-3D5AJ","outputId":"f7ac87a7-7de2-453c-bb9c-f9794a923884","execution":{"iopub.status.busy":"2023-06-19T20:42:03.0221Z","iopub.execute_input":"2023-06-19T20:42:03.022504Z","iopub.status.idle":"2023-06-19T20:42:03.04257Z","shell.execute_reply.started":"2023-06-19T20:42:03.022473Z","shell.execute_reply":"2023-06-19T20:42:03.0415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = \"/kaggle/input/airbus-ship-detection/train_v2/\"","metadata":{"id":"NWI3F3t2L33W","execution":{"iopub.status.busy":"2023-06-19T20:42:05.85532Z","iopub.execute_input":"2023-06-19T20:42:05.855725Z","iopub.status.idle":"2023-06-19T20:42:05.860298Z","shell.execute_reply.started":"2023-06-19T20:42:05.855667Z","shell.execute_reply":"2023-06-19T20:42:05.859117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check whether a given filename has a valid image file extension\ndef check_img(name):\n    return any(name.endswith(ext) for ext in ['.JPG','.jpg','.JPEG','.jpeg','.PNG','.png'])\n\n# Create an empty DataFrame\nimage_df = pd.DataFrame(columns=['Filename'])\n\n# Iterate over each file in the folder\nfor image_file in os.listdir(train_path):\n    if check_img(image_file):\n        # Append the image filename to a temporary DataFrame\n        temp_df = pd.DataFrame({'Filename': [image_file]})\n\n        # Concatenate the temporary DataFrame with the main DataFrame\n        image_df = pd.concat([image_df, temp_df], ignore_index=True)\n\n# Display the DataFrame\nprint(image_df)\n","metadata":{"id":"rQG720c-NCut","outputId":"35f60964-eb25-4538-ec9e-48f3f8e15871","execution":{"iopub.status.busy":"2023-06-19T20:42:11.007732Z","iopub.execute_input":"2023-06-19T20:42:11.00813Z","iopub.status.idle":"2023-06-19T20:46:15.349982Z","shell.execute_reply.started":"2023-06-19T20:42:11.008094Z","shell.execute_reply":"2023-06-19T20:46:15.348812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter the DataFrame to select rows with NA values in 'EncodedPixels' column\nimages_with_na = train_segmentation[train_segmentation['EncodedPixels'].isna()]\n\n# Retrieve the ImageIDs of images with NA values\nimage_ids_with_na = images_with_na['ImageId'].unique()\n\n# # Print the ImageIDs\n# for image_id in image_ids_with_na:\n#     print(image_id)","metadata":{"id":"5gglbwDpOK_O","execution":{"iopub.status.busy":"2023-06-19T20:46:58.130626Z","iopub.execute_input":"2023-06-19T20:46:58.131004Z","iopub.status.idle":"2023-06-19T20:46:58.192785Z","shell.execute_reply.started":"2023-06-19T20:46:58.13097Z","shell.execute_reply":"2023-06-19T20:46:58.191652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\n# Set the path to the new folder where the remaining image files will be saved\nnew_path = \"/kaggle/working/ResultImages/\"\n\n# Create the new folder if it doesn't exist\nif not os.path.exists(new_path):\n    os.makedirs(new_path)\n\n# Iterate over each image file in the train_path\nfor image_file in os.listdir(train_path):\n    # Extract the ImageID from the image filename\n    image_id = image_file.split('.')[0]  # Assuming the filenames have a format like \"ImageID.jpg\"\n\n    # Check if the ImageID is not in image_ids_with_na\n    if image_id not in image_ids_with_na:\n        # Create the full file path for the source file\n        source_path = os.path.join(train_path, image_file)\n\n        # Create the full file path for the destination file in the new folder\n        destination_path = os.path.join(new_path, image_file)\n\n        # Copy the file from the source path to the destination path\n        shutil.copyfile(source_path, destination_path)\n\n        print(f\"Copied file from: {source_path} to: {destination_path}\")","metadata":{"id":"bx6ZEHI6Rr-R","outputId":"4aeb26df-ba6c-4ee0-ef4e-bcf3ebc057f8","execution":{"iopub.status.busy":"2023-06-19T20:47:02.505059Z","iopub.execute_input":"2023-06-19T20:47:02.505884Z"},"trusted":true},"execution_count":null,"outputs":[]}]}