{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"[![Kaggle](https://kaggle.com/static/images/open-in-kaggle.svg)](https://kaggle.com/kernels/welcome?src=https://github.com/Eng-Dan/kaggle-happywhale-competition/blob/master/happywhale-dataframes.ipynb)\n\n# Context\nThis notebook generates two .csv files to use as input data for the happywhale 2022 competition.\n* simplified_train.csv\n* simplified_test.csv\n\nThe files contains the arrays retrieved from the resized train and test images dataset [Happywhale 2022 competition - Images 256 by 256](https://www.kaggle.com/datasets/engdan/happywhale-images-256-by-256).\n\nThe image arrays have been set to grayscale color map in order o reduce the size of the file. Thus, if your model should work with colored images, consider to use the original images from the competition.\n\n","metadata":{}},{"cell_type":"markdown","source":"# Packages and libraries","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n# Input data files are available in the read-only \"../input/\" directory\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Directory path variables","metadata":{}},{"cell_type":"code","source":"TRAIN_IMAGES_DIR = '../input/happywhale-images-256-by-256/resized_train_images'\nTEST_IMAGES_DIR = '../input/happywhale-images-256-by-256/resized_test_images'","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:10:24.541115Z","iopub.execute_input":"2022-03-18T15:10:24.541316Z","iopub.status.idle":"2022-03-18T15:10:24.547782Z","shell.execute_reply.started":"2022-03-18T15:10:24.541294Z","shell.execute_reply":"2022-03-18T15:10:24.546754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Working dataframes","metadata":{}},{"cell_type":"code","source":"train_dataset_csv = '../input/happy-whale-and-dolphin/train.csv'\nsample_submission_csv = '../input/happy-whale-and-dolphin/sample_submission.csv'","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:10:24.549065Z","iopub.execute_input":"2022-03-18T15:10:24.549294Z","iopub.status.idle":"2022-03-18T15:10:24.567695Z","shell.execute_reply.started":"2022-03-18T15:10:24.549263Z","shell.execute_reply":"2022-03-18T15:10:24.566511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_dataset_csv)\ntest_df = pd.read_csv(sample_submission_csv)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:10:24.569602Z","iopub.execute_input":"2022-03-18T15:10:24.570135Z","iopub.status.idle":"2022-03-18T15:10:24.743188Z","shell.execute_reply.started":"2022-03-18T15:10:24.570111Z","shell.execute_reply":"2022-03-18T15:10:24.741867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:10:24.744738Z","iopub.execute_input":"2022-03-18T15:10:24.744999Z","iopub.status.idle":"2022-03-18T15:10:24.764324Z","shell.execute_reply.started":"2022-03-18T15:10:24.744963Z","shell.execute_reply":"2022-03-18T15:10:24.762799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:10:24.765416Z","iopub.execute_input":"2022-03-18T15:10:24.765611Z","iopub.status.idle":"2022-03-18T15:10:24.775106Z","shell.execute_reply.started":"2022-03-18T15:10:24.765587Z","shell.execute_reply":"2022-03-18T15:10:24.774073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Retrieving the images array","metadata":{}},{"cell_type":"markdown","source":"The function bellow will add the `image_array` column to the required dataframes.","metadata":{}},{"cell_type":"code","source":"def add_image_array_column(dataframe, images_source_dir):\n    images_array = []\n    num_images = dataframe['image'].size\n    print('Process started...')\n    for index in range(num_images):\n        images_array.append(cv2.imread(os.path.join(images_source_dir, dataframe['image'].iloc[index]),\n                                       cv2.IMREAD_GRAYSCALE))\n    \n        if (index % 2000) == 0:\n            print('Images array added:', index, 'of', num_images)\n\n    dataframe['image_array'] = images_array\n    print('Process finished.')","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:10:24.776314Z","iopub.execute_input":"2022-03-18T15:10:24.777170Z","iopub.status.idle":"2022-03-18T15:10:24.788602Z","shell.execute_reply.started":"2022-03-18T15:10:24.777142Z","shell.execute_reply":"2022-03-18T15:10:24.787314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First, lets apply the function to the `train_df`.","metadata":{}},{"cell_type":"code","source":"add_image_array_column(train_df, TRAIN_IMAGES_DIR)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:10:24.790143Z","iopub.execute_input":"2022-03-18T15:10:24.790571Z","iopub.status.idle":"2022-03-18T15:15:34.024710Z","shell.execute_reply.started":"2022-03-18T15:10:24.790522Z","shell.execute_reply":"2022-03-18T15:15:34.023698Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, checking if all worked as expected.","metadata":{}},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:16:07.030757Z","iopub.execute_input":"2022-03-18T15:16:07.030973Z","iopub.status.idle":"2022-03-18T15:16:07.492672Z","shell.execute_reply.started":"2022-03-18T15:16:07.030951Z","shell.execute_reply":"2022-03-18T15:16:07.491647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_check = train_df.iloc[47696]\nprint(image_check.image)\nprint(image_check.species)\nprint(image_check.individual_id, '\\n')\nplt.imshow(image_check.image_array, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:16:49.691195Z","iopub.execute_input":"2022-03-18T15:16:49.691739Z","iopub.status.idle":"2022-03-18T15:16:49.850161Z","shell.execute_reply.started":"2022-03-18T15:16:49.691684Z","shell.execute_reply":"2022-03-18T15:16:49.849579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finally, applying the same process to the `test_df` (from `sample_submission.csv` file).","metadata":{}},{"cell_type":"code","source":"add_image_array_column(test_df, TEST_IMAGES_DIR)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:26:58.284145Z","iopub.execute_input":"2022-03-18T15:26:58.284399Z","iopub.status.idle":"2022-03-18T15:29:53.740297Z","shell.execute_reply.started":"2022-03-18T15:26:58.284368Z","shell.execute_reply":"2022-03-18T15:29:53.739265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CSV files generation","metadata":{}},{"cell_type":"code","source":"train_df.to_csv('simplified_train.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:48:09.344992Z","iopub.execute_input":"2022-03-18T15:48:09.345520Z","iopub.status.idle":"2022-03-18T15:48:16.699637Z","shell.execute_reply.started":"2022-03-18T15:48:09.345493Z","shell.execute_reply":"2022-03-18T15:48:16.698518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('simplified_test.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-18T15:48:27.639957Z","iopub.execute_input":"2022-03-18T15:48:27.640332Z","iopub.status.idle":"2022-03-18T15:48:31.574235Z","shell.execute_reply.started":"2022-03-18T15:48:27.640296Z","shell.execute_reply":"2022-03-18T15:48:31.573774Z"},"trusted":true},"execution_count":null,"outputs":[]}]}