{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\nimport glob\nimport shutil","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-19T22:53:36.696965Z","iopub.execute_input":"2023-06-19T22:53:36.697628Z","iopub.status.idle":"2023-06-19T22:53:36.705241Z","shell.execute_reply.started":"2023-06-19T22:53:36.697592Z","shell.execute_reply":"2023-06-19T22:53:36.704449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reading the segmentations of the train data\ntrain_segmentation = pd.read_csv(\"/kaggle/input/airbus-ship-detection/train_ship_segmentations_v2.csv\")\n#changing the EncodedPixels column to string\ntrain_segmentation['EncodedPixels'] = train_segmentation['EncodedPixels'].astype(\"string\")","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:53:39.355726Z","iopub.execute_input":"2023-06-19T22:53:39.356632Z","iopub.status.idle":"2023-06-19T22:53:39.888275Z","shell.execute_reply.started":"2023-06-19T22:53:39.356591Z","shell.execute_reply":"2023-06-19T22:53:39.887339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the information for the corrupted images specifically.\ncorrupted_images = ['6384c3e78.jpg']\ncorrupted_rows = train_segmentation[train_segmentation['ImageId'].isin(corrupted_images)]\ncorrupted_rows","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:53:42.456744Z","iopub.execute_input":"2023-06-19T22:53:42.457079Z","iopub.status.idle":"2023-06-19T22:53:42.484413Z","shell.execute_reply.started":"2023-06-19T22:53:42.457051Z","shell.execute_reply":"2023-06-19T22:53:42.483539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#deleting the corrupted image.\ntrain_segmentation = train_segmentation.drop(corrupted_rows.index)\ncorrupted_rows","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:53:45.446757Z","iopub.execute_input":"2023-06-19T22:53:45.447113Z","iopub.status.idle":"2023-06-19T22:53:45.49113Z","shell.execute_reply.started":"2023-06-19T22:53:45.447082Z","shell.execute_reply":"2023-06-19T22:53:45.490041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of rows in the data - {train_segmentation.shape[0]}')","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:53:48.127166Z","iopub.execute_input":"2023-06-19T22:53:48.127512Z","iopub.status.idle":"2023-06-19T22:53:48.132897Z","shell.execute_reply.started":"2023-06-19T22:53:48.127481Z","shell.execute_reply":"2023-06-19T22:53:48.13193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculating the number of unique images in the 'ImageId' column of the 'train_segmentation'\ntrain_images_number = train_segmentation['ImageId'].nunique()\nprint(f'Number of Train Images - {train_images_number}')","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:53:50.827986Z","iopub.execute_input":"2023-06-19T22:53:50.828329Z","iopub.status.idle":"2023-06-19T22:53:50.887868Z","shell.execute_reply.started":"2023-06-19T22:53:50.8283Z","shell.execute_reply":"2023-06-19T22:53:50.88692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculate the number of images in 'train_segmentation' that do not have ship annotations\nimages_without_ships = train_segmentation['EncodedPixels'].isna().sum()\nprint(f'Number of images without ships - {images_without_ships}')","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:53:57.763475Z","iopub.execute_input":"2023-06-19T22:53:57.763865Z","iopub.status.idle":"2023-06-19T22:53:57.782575Z","shell.execute_reply.started":"2023-06-19T22:53:57.763835Z","shell.execute_reply":"2023-06-19T22:53:57.781571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = \"/kaggle/input/airbus-ship-detection/train_v2/\"\n\n#check whether a given filename has a valid image file extension\n# def check_img(name):\n#     return any(name.endswith(ext) for ext in ['.JPG','.jpg','.JPEG','.jpeg','.PNG','.png'])\n\n# # Create an empty DataFrame\n# image_df = pd.DataFrame(columns=['Filename'])\n\n# # Iterate over each file in the folder\n# for image_file in os.listdir(train_path):\n#     if check_img(image_file):\n#         # Append the image filename to a temporary DataFrame\n#         temp_df = pd.DataFrame({'Filename': [image_file]})\n\n#         # Concatenate the temporary DataFrame with the main DataFrame\n#         image_df = pd.concat([image_df, temp_df], ignore_index=True)\n\n# # Display the DataFrame\n# print(image_df)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:54:10.254523Z","iopub.execute_input":"2023-06-19T22:54:10.25488Z","iopub.status.idle":"2023-06-19T22:54:10.260345Z","shell.execute_reply.started":"2023-06-19T22:54:10.25485Z","shell.execute_reply":"2023-06-19T22:54:10.259255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter the DataFrame to select rows with NA values in 'EncodedPixels' column\nimages_with_na = train_segmentation[train_segmentation['EncodedPixels'].isna()]\n\n# Retrieve the ImageIDs of images with NA values\nimage_ids_with_na = images_with_na['ImageId'].unique()\n\n# Print the ImageIDs\nfor image_id in image_ids_with_na:\n    print(image_id)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:54:15.869825Z","iopub.execute_input":"2023-06-19T22:54:15.870205Z","iopub.status.idle":"2023-06-19T22:54:16.764078Z","shell.execute_reply.started":"2023-06-19T22:54:15.870175Z","shell.execute_reply":"2023-06-19T22:54:16.76222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df = pd.DataFrame(columns=['ImageID'])\n\n# Iterate over each image file in the train_path\nfor image_file in os.listdir(train_path):\n\n    # Check if the ImageID is not in image_ids_with_na\n    if image_file not in image_ids_with_na:\n        # Append the ImageID to the DataFrame\n        temp_df = pd.DataFrame({'ImageID': [image_file]})\n        \n        # Concatenate the temporary DataFrame with the main DataFrame\n        image_df = pd.concat([image_df, temp_df], ignore_index=True)\n\n# Display the DataFrame\nprint(image_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T22:54:36.083344Z","iopub.execute_input":"2023-06-19T22:54:36.083724Z","iopub.status.idle":"2023-06-19T23:19:12.837903Z","shell.execute_reply.started":"2023-06-19T22:54:36.083689Z","shell.execute_reply":"2023-06-19T23:19:12.836828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\n# Set the path to the new folder where the remaining image files will be saved\nnew_path = \"/kaggle/working/ResultImages/\"\n\n# Create the new folder if it doesn't exist\nif not os.path.exists(new_path):\n    os.makedirs(new_path)\n\n# Iterate over each image file in the train_path\nfor image_file in os.listdir(train_path):\n    # Extract the ImageID from the image filename\n\n    # Check if the ImageID is not in image_ids_with_na\n    if image_file not in image_ids_with_na:\n        # Create the full file path for the source file\n        source_path = os.path.join(train_path, image_file)\n\n        # Create the full file path for the destination file in the new folder\n        destination_path = os.path.join(new_path, image_file)\n\n        # Copy the file from the source path to the destination path\n        shutil.copyfile(source_path, destination_path)\n\n        print(f\"Copied file from: {source_path} to: {destination_path}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-19T23:28:26.130123Z","iopub.execute_input":"2023-06-19T23:28:26.13048Z"},"trusted":true},"execution_count":null,"outputs":[]}]}