{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom scipy.stats import entropy\nimport matplotlib.pyplot as plt\n\nplt.rcParams[\"figure.figsize\"] = (10,10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data_df = pd.read_csv(\"/kaggle/input/hotel-id-2021-fgvc8/train.csv\")\ntrain_data_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of images in the dataset:\",train_data_df.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Checking for duplicates\nduplicate_images = train_data_df[train_data_df.duplicated(subset=['image'])==True]['image'].values\nfor dup in duplicate_images:\n    print(train_data_df[train_data_df['image']==dup])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Two duplicates found. Most probably by the same user. As chain, hotel_id and timestamp are identical. "},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check if nan values present\nprint(\"Number of NaN values present:\",train_data_df.isna().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Chain value 0 represents individual hotels\nprint(\"Number of Unique Hotel Chains:\",train_data_df['chain'].nunique()-1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Hotel Chains and how many hotel each hotel chain contain\nhotel_count = {}\nfor hotel_chain_id in train_data_df['chain'].unique():\n    key = hotel_chain_id\n    value = train_data_df[train_data_df['chain']==hotel_chain_id]['hotel_id'].nunique()\n    hotel_count[key] = value\n\n#hotel_count.pop(0)\nbar = plt.bar(x=hotel_count.keys(),height=hotel_count.values(),color=\"blueviolet\")\nplt.xlabel(\"Hotel Chain ID\")\nplt.ylabel(\"Count\")\nplt.title(\"Hotel Chains and their hotel counts\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Lots of individual hotels in the data."},{"metadata":{"trusted":true},"cell_type":"code","source":"# Number of hotels and how many images for each hotel\nhotels = train_data_df['hotel_id'].unique()\nhotels_image_count = []\nfor hotel in hotels:\n    cnt = train_data_df[train_data_df['hotel_id']==hotel]['image'].nunique()\n    hotels_image_count.append(cnt)\n\nhotel_image_df = pd.DataFrame({\"hotel_id\":map(str,hotels),\"image_count\":hotels_image_count})\nhotel_image_df.sort_values(by=\"image_count\",ascending=False,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=[15,15])\ntop_50_hotel_image_df = hotel_image_df.iloc[:50,:]\nplt.bar(x=top_50_hotel_image_df[\"hotel_id\"],height=top_50_hotel_image_df[\"image_count\"],color=\"blueviolet\")\nplt.xlabel(\"Hotel ID\")\nplt.xticks(rotation=45)\nplt.ylabel(\"Image Count\")\nplt.title(\"Hotel and their image count (Top 50)\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=[15,15])\nbottom_50_hotel_image_df = hotel_image_df.iloc[-50:,:]\nplt.bar(x=bottom_50_hotel_image_df[\"hotel_id\"],height=bottom_50_hotel_image_df[\"image_count\"],color=\"blueviolet\")\nplt.xlabel(\"Hotel ID\")\nplt.xticks(rotation=45)\nplt.ylabel(\"Image Count\")\nplt.title(\"Hotel and their image count (Bottom 50)\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Huge imbalance between top 50 and bottom 50. A quanitification will show a better picture"},{"metadata":{"trusted":true},"cell_type":"code","source":"def shannon_entropy(no_of_classes,sizes,dataset_size):\n    sh_en = 0\n    for i in range(no_of_classes):\n        sh_en += (sizes[i]/dataset_size)*np.log(sizes[i]/dataset_size)\n    return -sh_en\n\ndef quant_imbalance(no_of_classes,sizes,dataset_size):\n    sh_en = shannon_entropy(no_of_classes,sizes,dataset_size)\n    return sh_en/no_of_classes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Imbalance quantification using Shannon Entropy\nprint(quant_imbalance(hotel_image_df.shape[0],hotel_image_df['image_count'].values.tolist(),hotel_image_df.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This number quantifies how badly the data is distributed. Proper measures will have to be taken while training to avoid overfitting to a few classes. "}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}