{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import code needed for dataset exploration and model training ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport PIL\nimport tensorflow as tf\nimport random\nfrom IPython.core.debugger import set_trace\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import Sequential, layers\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-28T16:11:37.251432Z","iopub.execute_input":"2022-03-28T16:11:37.251962Z","iopub.status.idle":"2022-03-28T16:11:44.963511Z","shell.execute_reply.started":"2022-03-28T16:11:37.251842Z","shell.execute_reply":"2022-03-28T16:11:44.962675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pwd","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:11:44.96594Z","iopub.execute_input":"2022-03-28T16:11:44.966652Z","iopub.status.idle":"2022-03-28T16:11:44.976956Z","shell.execute_reply.started":"2022-03-28T16:11:44.966602Z","shell.execute_reply":"2022-03-28T16:11:44.975909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:11:44.978394Z","iopub.execute_input":"2022-03-28T16:11:44.978826Z","iopub.status.idle":"2022-03-28T16:11:45.717553Z","shell.execute_reply.started":"2022-03-28T16:11:44.978779Z","shell.execute_reply":"2022-03-28T16:11:45.716554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Let's create out dataframe with pictures and masks**","metadata":{}},{"cell_type":"markdown","source":"# First we create a global dataframe with all picture files in our data folder","metadata":{}},{"cell_type":"code","source":"df = pd.DataFrame(columns=['directory','hotel_id', 'image_id', 'image_width', 'image_height','image_size'])","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:11:45.720617Z","iopub.execute_input":"2022-03-28T16:11:45.721427Z","iopub.status.idle":"2022-03-28T16:11:45.736711Z","shell.execute_reply.started":"2022-03-28T16:11:45.721359Z","shell.execute_reply":"2022-03-28T16:11:45.735751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfor dirname, _, filenames in os.walk('/kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9/'):\n    for filename in filenames: \n        try:\n#             set_trace()\n            if '/kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9/train_images/' in dirname :\n                hotel_id = dirname.replace('/kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9/train_images/','')\n            else:\n                hotel_id = None\n            image_size=PIL.Image.open(os.path.join(dirname, filename)).size\n            row = pd.DataFrame({'directory':dirname,'hotel_id':hotel_id, 'image_id':filename, 'image_width':[image_size[0]], 'image_height':[image_size[1]],'image_size':[image_size]})\n            df = pd.concat([df,row])\n            print(hotel_id,filename,image_size)\n        except:\n            pass","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-03-28T16:26:01.031955Z","iopub.execute_input":"2022-03-28T16:26:01.032304Z","iopub.status.idle":"2022-03-28T16:40:54.472719Z","shell.execute_reply.started":"2022-03-28T16:26:01.032264Z","shell.execute_reply":"2022-03-28T16:40:54.471892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pwd","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:40:54.474388Z","iopub.execute_input":"2022-03-28T16:40:54.475253Z","iopub.status.idle":"2022-03-28T16:40:54.480489Z","shell.execute_reply.started":"2022-03-28T16:40:54.475215Z","shell.execute_reply":"2022-03-28T16:40:54.479739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:40:54.481626Z","iopub.execute_input":"2022-03-28T16:40:54.481839Z","iopub.status.idle":"2022-03-28T16:40:55.201719Z","shell.execute_reply.started":"2022-03-28T16:40:54.481814Z","shell.execute_reply":"2022-03-28T16:40:55.200489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:40:55.205723Z","iopub.execute_input":"2022-03-28T16:40:55.20599Z","iopub.status.idle":"2022-03-28T16:40:55.2178Z","shell.execute_reply.started":"2022-03-28T16:40:55.205946Z","shell.execute_reply":"2022-03-28T16:40:55.216675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(columns=['index'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:40:55.221104Z","iopub.execute_input":"2022-03-28T16:40:55.221395Z","iopub.status.idle":"2022-03-28T16:40:55.267263Z","shell.execute_reply.started":"2022-03-28T16:40:55.221359Z","shell.execute_reply":"2022-03-28T16:40:55.266363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(df.tail(1).index,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:40:55.268566Z","iopub.execute_input":"2022-03-28T16:40:55.26891Z","iopub.status.idle":"2022-03-28T16:40:55.292242Z","shell.execute_reply.started":"2022-03-28T16:40:55.268861Z","shell.execute_reply":"2022-03-28T16:40:55.290749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:40:55.293938Z","iopub.execute_input":"2022-03-28T16:40:55.29473Z","iopub.status.idle":"2022-03-28T16:40:55.326879Z","shell.execute_reply.started":"2022-03-28T16:40:55.294682Z","shell.execute_reply":"2022-03-28T16:40:55.325733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('df.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:40:55.328362Z","iopub.execute_input":"2022-03-28T16:40:55.330592Z","iopub.status.idle":"2022-03-28T16:40:55.809299Z","shell.execute_reply.started":"2022-03-28T16:40:55.330004Z","shell.execute_reply":"2022-03-28T16:40:55.808604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Then we can create our image dataframe containing all the unmasked pictures","metadata":{}},{"cell_type":"code","source":"image_df = df[df['hotel_id'].notnull()]","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:43:21.375818Z","iopub.execute_input":"2022-03-28T16:43:21.376194Z","iopub.status.idle":"2022-03-28T16:43:21.403278Z","shell.execute_reply.started":"2022-03-28T16:43:21.376154Z","shell.execute_reply":"2022-03-28T16:43:21.402241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:43:21.679912Z","iopub.execute_input":"2022-03-28T16:43:21.680187Z","iopub.status.idle":"2022-03-28T16:43:21.699088Z","shell.execute_reply.started":"2022-03-28T16:43:21.680156Z","shell.execute_reply":"2022-03-28T16:43:21.698258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df.to_csv('image_df.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-28T16:43:33.453981Z","iopub.execute_input":"2022-03-28T16:43:33.454985Z","iopub.status.idle":"2022-03-28T16:43:33.825074Z","shell.execute_reply.started":"2022-03-28T16:43:33.454947Z","shell.execute_reply":"2022-03-28T16:43:33.824467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We are able to analyze the data from images :**","metadata":{}},{"cell_type":"markdown","source":"**Data about hotel chains**","metadata":{}},{"cell_type":"code","source":"chains = image_df.groupby('hotel_id').size()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:04:25.868256Z","iopub.execute_input":"2022-03-28T17:04:25.86881Z","iopub.status.idle":"2022-03-28T17:04:25.889322Z","shell.execute_reply.started":"2022-03-28T17:04:25.868775Z","shell.execute_reply":"2022-03-28T17:04:25.888278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chains.sort_values(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:04:26.129794Z","iopub.execute_input":"2022-03-28T17:04:26.130286Z","iopub.status.idle":"2022-03-28T17:04:26.135625Z","shell.execute_reply.started":"2022-03-28T17:04:26.130237Z","shell.execute_reply":"2022-03-28T17:04:26.134592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('we have a total of',chains.count(),'hotel chains, and there is an average of', round(chains.mean()), 'pictures per chain, but as it is a skewed representation, we should consider the median which is',chains.median())","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:04:26.31115Z","iopub.execute_input":"2022-03-28T17:04:26.311636Z","iopub.status.idle":"2022-03-28T17:04:26.31776Z","shell.execute_reply.started":"2022-03-28T17:04:26.311602Z","shell.execute_reply":"2022-03-28T17:04:26.31698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nhotel_id_size = sns.barplot(x=chains.index,y=chains.values)\nhotel_id_size.set_title('Number of pictures available per hotel chain')\nhotel_id_size.set_xlabel('Hotel chain')\nhotel_id_size.set_ylabel('Number of Pictures')\nplt.ylim(0,100)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-03-28T17:04:26.505968Z","iopub.execute_input":"2022-03-28T17:04:26.506461Z","iopub.status.idle":"2022-03-28T17:05:10.936337Z","shell.execute_reply.started":"2022-03-28T17:04:26.506426Z","shell.execute_reply":"2022-03-28T17:05:10.935468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data about picture sizes**","metadata":{}},{"cell_type":"code","source":"picture_sizes = image_df.groupby('image_size').size()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.968304Z","iopub.status.idle":"2022-03-28T17:05:10.96885Z","shell.execute_reply.started":"2022-03-28T17:05:10.968638Z","shell.execute_reply":"2022-03-28T17:05:10.968665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"picture_sizes.sort_values(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.970226Z","iopub.status.idle":"2022-03-28T17:05:10.970788Z","shell.execute_reply.started":"2022-03-28T17:05:10.970505Z","shell.execute_reply":"2022-03-28T17:05:10.970551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"picture_sizes","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.972723Z","iopub.status.idle":"2022-03-28T17:05:10.973366Z","shell.execute_reply.started":"2022-03-28T17:05:10.973084Z","shell.execute_reply":"2022-03-28T17:05:10.973113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('we have a total of',picture_sizes.count(),'picture sizes, the most represented size is', picture_sizes.tail(1).index[0] )","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.97486Z","iopub.status.idle":"2022-03-28T17:05:10.975672Z","shell.execute_reply.started":"2022-03-28T17:05:10.975377Z","shell.execute_reply":"2022-03-28T17:05:10.975406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\npicture_shape = sns.barplot(x=picture_sizes.index,y=picture_sizes.values)\npicture_shape.set_title('Picture size population')\npicture_shape.set_xlabel('Picture size')\npicture_shape.set_ylabel('Number of Pictures')","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.97739Z","iopub.status.idle":"2022-03-28T17:05:10.977943Z","shell.execute_reply.started":"2022-03-28T17:05:10.977718Z","shell.execute_reply":"2022-03-28T17:05:10.977747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_df = df[df['hotel_id'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.978914Z","iopub.status.idle":"2022-03-28T17:05:10.979247Z","shell.execute_reply.started":"2022-03-28T17:05:10.979078Z","shell.execute_reply":"2022-03-28T17:05:10.9791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_df.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.980859Z","iopub.status.idle":"2022-03-28T17:05:10.981177Z","shell.execute_reply.started":"2022-03-28T17:05:10.981003Z","shell.execute_reply":"2022-03-28T17:05:10.981025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_df","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.982618Z","iopub.status.idle":"2022-03-28T17:05:10.98293Z","shell.execute_reply.started":"2022-03-28T17:05:10.982763Z","shell.execute_reply":"2022-03-28T17:05:10.982785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_df.to_csv('mask_df.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.983908Z","iopub.status.idle":"2022-03-28T17:05:10.984238Z","shell.execute_reply.started":"2022-03-28T17:05:10.984072Z","shell.execute_reply":"2022-03-28T17:05:10.984095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_df.groupby('image_size').size().sort_values()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.986118Z","iopub.status.idle":"2022-03-28T17:05:10.986435Z","shell.execute_reply.started":"2022-03-28T17:05:10.986269Z","shell.execute_reply":"2022-03-28T17:05:10.986291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('we have a total of',mask_df.groupby('image_size').size().sort_values().count(),'mask sizes, the most represented size is', mask_df.groupby('image_size').size().sort_values().tail(1).index[0] )","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.98775Z","iopub.status.idle":"2022-03-28T17:05:10.988063Z","shell.execute_reply.started":"2022-03-28T17:05:10.987892Z","shell.execute_reply":"2022-03-28T17:05:10.987913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**After data analysis, we can see that we have some problems to deal with in our later preprocessing, modeling and training :**\n\n* Unbalanced dataset : Some Hotel chains have much more pictures than others and there is a risk that the model learns more from these chains\n\n* Image ratio variety : We find a wide variety of image sizes and image ratios which makes it difficult to resize all images to have the same shape\n\n* Mask size variety : I am not sure how the masks are meant to be used but we have a wide range of masks sizes and ratio.\n\n**In the next steps we will try to :**\n\n* Reduce image size but keep the same image ratio\n\n* Use data augmentation on the complete dataset and try to resize as (256,256)\n\n* Implement random mask for each picture during the data augmentation process","metadata":{}},{"cell_type":"code","source":"pwd","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.989655Z","iopub.status.idle":"2022-03-28T17:05:10.989971Z","shell.execute_reply.started":"2022-03-28T17:05:10.989804Z","shell.execute_reply":"2022-03-28T17:05:10.989827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs('hotel_id_dataset_512x512',exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.991218Z","iopub.status.idle":"2022-03-28T17:05:10.991544Z","shell.execute_reply.started":"2022-03-28T17:05:10.991357Z","shell.execute_reply":"2022-03-28T17:05:10.991379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_image(dirname,filename):\n    max_size=(512,512)\n    picture=PIL.Image.open(os.path.join(dirname,filename))\n    cover = mask_df.iloc[random.randint(0,4949)]\n    mask=PIL.Image.open(os.path.join(cover['directory'],cover['image_id']))\n    if picture.width > picture.height:\n         picture=picture.rotate(90,expand=True)\n    if mask.width > mask.height:\n        mask=mask.rotate(90,expand=True)\n    picture.thumbnail(max_size) \n    mask.thumbnail(max_size)\n    picture.paste(mask,(0,0),mask)\n    new_filename = 'preproc_'+filename\n    new_dirname = dirname.replace('/kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9/','/kaggle/working/hotel_id_dataset_512x512/')\n    os.makedirs(new_dirname,exist_ok=True)\n    picture.save(os.path.join(new_dirname,new_filename))\n    return picture","metadata":{"execution":{"iopub.status.busy":"2022-03-28T17:05:10.993021Z","iopub.status.idle":"2022-03-28T17:05:10.993494Z","shell.execute_reply.started":"2022-03-28T17:05:10.993284Z","shell.execute_reply":"2022-03-28T17:05:10.993313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfor dirname, _, filenames in os.walk('/kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9'):\n    for filename in filenames:\n        try:\n            if '/kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9/train_images/' in dirname and '.jpg' in filename:\n                  prepare_image(dirname,filename)\n        except:\n            pass","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We have now a complete set of masked images with max size of 512x512 but keeping aspect ratio","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}