{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib\nimport matplotlib.pyplot as plt\nimport matplotlib.pylab as pylab\nimport seaborn as sns\n\n%matplotlib inline\nmatplotlib.style.use(\"ggplot\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read csv data","metadata":{}},{"cell_type":"code","source":"data_folder = \"/kaggle/input/hotel-id-2021-fgvc8/\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(data_folder + \"train.csv\", parse_dates=[\"timestamp\"])\nsubmission_df = pd.read_csv(data_folder + \"sample_submission.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Look at csv data","metadata":{}},{"cell_type":"markdown","source":"## Train data","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of records: {}\".format(len(train_df)))\nprint(\"Number of images: {}\".format(train_df[\"image\"].unique().size))\nprint(\"Number of chains: {}\".format(train_df[\"chain\"].unique().size))\nprint(\"Number of hotels: {}\".format(train_df[\"hotel_id\"].unique().size))\nprint(\"Newest image: {}\".format(train_df[\"timestamp\"].max()))\nprint(\"Oldest image: {}\".format(train_df[\"timestamp\"].min()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There seems to be some duplicated images so let's look at them","metadata":{}},{"cell_type":"code","source":"train_df[train_df[\"image\"].duplicated(keep=False)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Doesn't look that bad, just 2 duplicates but they belong to same hotel so we can simple drop them","metadata":{}},{"cell_type":"code","source":"train_df = train_df.drop_duplicates(subset=[\"image\"], keep=\"first\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at the chain values per hotel, a hotel should belong to one chain","metadata":{}},{"cell_type":"code","source":"group_df = train_df.groupby(\"hotel_id\").agg({\"chain\": [pd.Series.nunique, pd.Series.unique, \"max\"]})\ngroup_df.columns = [\"_\".join(x) for x in group_df.columns.ravel()]\ngroup_df.sort_values(\"chain_nunique\")[::-1].head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are some hotels that belong to multiple chains though\n","metadata":{}},{"cell_type":"code","source":"hotels_df = train_df[train_df[\"hotel_id\"].isin(group_df[group_df[\"chain_nunique\"] > 1].index)]\nhotels_group_df = hotels_df.groupby([\"hotel_id\", \"chain\"]).size().to_frame(\"image_count\").reset_index()\n\nfig = px.bar(hotels_group_df, x=\"hotel_id\", y=\"image_count\", color=hotels_group_df[\"chain\"].astype(str))\nfig.update_xaxes(title_text=\"Hotel ID\", type=\"category\")\nfig.update_yaxes(title_text=\"Image count\")\nfig.update_layout(title=\"Image count per hotel and chain\", legend=dict(title=\"Chain\"))\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"which is explained [here](https://www.kaggle.com/c/hotel-id-2021-fgvc8/discussion/230768#1264813) by the host:\n> During the process of prepping the dataset, we discovered a few hotels that needed to be merged -- that is, there were two different IDs that referred to the same hotel. It would seem that when we did this merging, the chain ended up getting both the \"unknown\" label (0) and the correct label (whatever the non-zero chain number is). So you can use the non-zero label!\n\nSo we can fix the data accordingly by taking the non zero chain","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(data_folder + \"train.csv\", parse_dates=[\"timestamp\"])\ngroup_df = train_df.groupby(\"hotel_id\").agg({\"chain\": \"max\"}).reset_index()\ntrain_df = train_df.merge(group_df[[\"hotel_id\", \"chain\"]], on=\"hotel_id\", suffixes=(\"\", \"_new\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And check that the data is fixed","metadata":{}},{"cell_type":"code","source":"group_df = train_df.groupby(\"hotel_id\").agg({\"chain\": [pd.Series.nunique, pd.Series.unique], \"chain_new\": [pd.Series.nunique, pd.Series.unique]})\ngroup_df.sort_values((\"chain\", \"nunique\"))[::-1].head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks good so we can replace the old chain column with the new chain","metadata":{}},{"cell_type":"code","source":"train_df[\"chain\"] = train_df[\"chain_new\"]\ntrain_df.drop(columns=[\"chain_new\"], inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"submission_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Chains","metadata":{}},{"cell_type":"code","source":"chain_group_df = train_df.groupby([\"chain\"]).agg({\"hotel_id\": [pd.Series.nunique], \"image\" : [pd.Series.nunique]})\nchain_group_df.columns = [\"_\".join(x) for x in chain_group_df.columns.ravel()]\nchain_group_df = chain_group_df.reset_index().sort_values(\"hotel_id_nunique\")[::-1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig = make_subplots(rows=2, cols=1, vertical_spacing=0.02, shared_xaxes=True,)\n\n# fig.add_trace(go.Bar(x=group_df[\"chain\"].astype(str), y=group_df[\"hotel_id_nunique\"], showlegend = False, name=\"Hotel count\"), 1, 1)\n# fig.add_trace(go.Bar(x=group_df[\"chain\"].astype(str), y=group_df[\"image_nunique\"], showlegend = False, name=\"Image count\"), 2, 1)\n\n# fig.update_yaxes(title_text=\"Hotel count\", row=1, col=1)\n# fig.update_yaxes(title_text=\"Image count\", row=2, col=1)\n# fig.update_xaxes(title_text=\"Hotel ID\", row=2, col=1)\n# fig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(chain_group_df, x=\"chain\", y=\"hotel_id_nunique\",\n                 size=\"image_nunique\", color = \"image_nunique\",\n                 hover_name = None,\n                 log_y=True, size_max=75)\n\nfig.update_yaxes(title_text=\"Hotel count\")\nfig.update_xaxes(title_text=\"Chain ID\")\nfig.update_layout(title=\"Hotel and image count per chain\", coloraxis=dict(colorbar=dict(title=\"Image count\")))\nfig.update_traces(hovertemplate=\"Chain: %{x} <br>Hotel count: %{y}<br>Image count: %{marker.size}\")\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group_df = train_df.groupby([\"chain\", \"hotel_id\"]).size().to_frame(\"image_count\").reset_index()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"big_chains = chain_group_df[chain_group_df[\"hotel_id_nunique\"] >= 75][\"chain\"].values\nmid_chains = chain_group_df[(chain_group_df[\"hotel_id_nunique\"] < 75) & (chain_group_df[\"hotel_id_nunique\"] >= 10)][\"chain\"].values\nsmall_chains = chain_group_df[chain_group_df[\"hotel_id_nunique\"] < 10][\"chain\"].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.box(group_df[group_df[\"chain\"].isin(big_chains)], x=\"chain\", y=\"image_count\", height=350)\nfig.update_xaxes(title_text=\"Chain ID\", type=\"category\")\nfig.update_layout(title=\"Image count per hotel: Big chains (75 hotels and more)\")\nfig.show()\n\nfig = px.box(group_df[group_df[\"chain\"].isin(mid_chains)], x=\"chain\", y=\"image_count\", height=350)\nfig.update_xaxes(title_text=\"Chain ID\", type=\"category\")\nfig.update_layout(title=\"Image count per hotel: Mid chains (10-75 hotels)\")\nfig.show()\n\nfig = px.box(group_df[group_df[\"chain\"].isin(small_chains)], x=\"chain\", y=\"image_count\", height=350)\nfig.update_xaxes(title_text=\"Chain ID\", type=\"category\")\nfig.update_layout(title=\"Image count per hotel: Small chains (less than 10 hotels)\")\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hotels","metadata":{}},{"cell_type":"code","source":"group_df = train_df.groupby([\"hotel_id\"]).size().to_frame(\"image_count\").sort_values(\"image_count\")[::-1].reset_index()\n\n# top and low\nlow_df = group_df.iloc[-50:]\ntop_df = group_df.iloc[:50]\n\nfig = make_subplots(rows=2, cols=2, \n                    specs=[[{\"colspan\": 2}, None], [{}, {}]],\n                    horizontal_spacing=0.02, vertical_spacing=0.2, \n                    shared_yaxes=True,\n                    subplot_titles=(\"\", \"Top 50\", \"Bottom 50\"))\n\n\nfig.add_trace(go.Scatter(x=group_df[\"hotel_id\"], y=group_df[\"image_count\"], showlegend = False), 1, 1)\nfig.add_trace(go.Bar(x=top_df[\"hotel_id\"], y=top_df[\"image_count\"], showlegend = False), 2, 1)\nfig.add_trace(go.Bar(x=low_df[\"hotel_id\"], y=low_df[\"image_count\"], showlegend = False), 2, 2)\n\nfig.update_yaxes(title_text=\"Image count\", row=1, col=1)\nfig.update_yaxes(title_text=\"Image count\", row=2, col=1)\nfig.update_xaxes(type=\"category\", visible=False, row=1, col=1)\nfig.update_xaxes(title_text=\"Hotel ID\", type=\"category\", row=2, col=1)\nfig.update_xaxes(title_text=\"Hotel ID\", type=\"category\", row=2, col=2)\n\nfig.update_layout(title=\"Image count per hotel\", height=550)\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(group_df, x=\"image_count\", nbins=25, marginal=\"box\", height=500)\nfig.update_layout(title=\"Distribution of image count per hotel\")\nfig.update_traces(hovertemplate=\"Image count: %{x} <br>Hotel count: %{y}\")\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Timestamps","metadata":{}},{"cell_type":"code","source":"group_df = train_df.groupby([train_df[\"timestamp\"].dt.to_period(\"M\")])[\"image\"].count().reset_index()\n\nfig = px.bar(group_df, x=group_df[\"timestamp\"].astype(str), y=\"image\")\nfig.update_yaxes(title_text=\"Image count\")\nfig.update_xaxes(title_text=\"Time\", type=\"category\")\nfig.update_layout(title=\"Image count by months\", height=350)\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hotel_df = train_df[train_df[\"hotel_id\"] == 53586]\ngroup_df = hotel_df.groupby([hotel_df[\"timestamp\"].dt.to_period(\"M\")])[\"image\"].count().reset_index()\n\nfig = px.bar(group_df, x=group_df[\"timestamp\"].astype(str), y=\"image\")\nfig.update_yaxes(title_text=\"Image count\")\nfig.update_xaxes(title_text=\"Time\", type=\"category\")\nfig.update_layout(title=\"Image count by months for hotel 53586\", height=350)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group_df = train_df.groupby([train_df[\"timestamp\"].dt.to_period(\"M\"), \"chain\"])[\"image\"].count().reset_index()\n\nfig = px.scatter(group_df, x=group_df[\"timestamp\"].astype(str), y=\"chain\",\n                 size=\"image\", color = \"image\",\n                 hover_name = None,\n                 size_max=25)\n\nfig.update_yaxes(title_text=\"Chain\", type=\"category\")\nfig.update_xaxes(title_text=\"Time\")\nfig.update_layout(title=\"Image count by months for chains\", coloraxis=dict(colorbar=dict(title=\"Image count\")))\nfig.update_traces(hovertemplate=\"Time: %{x} <br>Chain: %{y}<br>Image count: %{marker.size}\")\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chain_df = train_df[train_df[\"chain\"] == 6]\ngroup_df = chain_df.groupby([chain_df[\"timestamp\"].dt.to_period(\"M\"), \"hotel_id\"])[\"image\"].count().reset_index()\n\nfig = px.scatter(group_df, x=group_df[\"timestamp\"].astype(str), y=\"hotel_id\",\n                 size=\"image\", color = \"image\",\n                 hover_name = None,\n                 size_max=25)\n\nfig.update_yaxes(title_text=\"Hotel\", type=\"category\")\nfig.update_xaxes(title_text=\"Time\")\nfig.update_layout(title=\"Image count by months for hotels of chain 6\", coloraxis=dict(colorbar=dict(title=\"Image count\")))\nfig.update_traces(hovertemplate=\"Time: %{x} <br>Hotel: %{y}<br>Image count: %{marker.size}\")\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Look at images","metadata":{}},{"cell_type":"code","source":"from PIL import Image","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def open_image(row_df):\n    return Image.open(f\"{data_folder}train_images/{row_df.chain.astype(int)}/{row_df.image}\")\n\n\ndef select_data(data_df, chain_id, hotel_id, N):   \n    if hotel_id is not None:\n        sub_df = data_df[data_df[\"hotel_id\"] == hotel_id]\n    elif chain_id is not None:\n        sub_df = data_df[data_df[\"chain\"] == chain_id]\n    else:\n        sub_df = data_df\n        \n    if N is not None:\n        sub_df = sub_df.sample(N)\n    \n    return sub_df\n\n\ndef show_images(data_df, nrows=None, ncols=None, fig_title=None):\n    N = len(data_df)\n    \n    if nrows is None:\n        nrows = 1\n    if ncols is None:\n        ncols = N\n    \n    fig, axs = plt.subplots(nrows, ncols, figsize=(24,int(5*nrows)))\n    if not isinstance(axs, np.ndarray):\n        axs = np.array(axs)\n        \n    axs = axs.ravel()\n    \n    for i in range(0, N):\n        row_df = data_df.iloc[i]\n        image = open_image(row_df)\n        axs[i].imshow(image)\n        axs[i].set_title(f\"{row_df.chain}:{row_df.hotel_id}:{row_df.image}\\n\" + \n                          f\"Time: {row_df.timestamp}\\n\" + \n                          f\"Size: {np.shape(image)}\")\n        axs[i].axis(\"off\")\n        \n    if fig_title is not None:\n        fig.suptitle(fig_title, fontsize=16)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random sample of 10 images","metadata":{}},{"cell_type":"code","source":"sample_df = select_data(train_df, None, None, 10)\nshow_images(sample_df, 2, 5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5 sample images of hotel 48897","metadata":{}},{"cell_type":"code","source":"sample_df = select_data(train_df, None, 48897, 5)\nshow_images(sample_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check image sizes","metadata":{}},{"cell_type":"code","source":"N = 500","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = select_data(train_df, None, None, N)\nx_array = []\ny_array = []\nz_array = []\n\nfor i in range(0, N):\n    I = open_image(sample_df.iloc[i])\n    x, y, z = np.shape(I)\n    x_array.append(x)\n    y_array.append(y)\n    z_array.append(z)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(data={\"x\": x_array, \"y\": y_array}).describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\nfig.add_trace(go.Box(x=z_array, name=\"Z\", boxpoints=\"all\"))\nfig.add_trace(go.Box(x=y_array, name=\"Y\", boxpoints=\"all\"))\nfig.add_trace(go.Box(x=x_array, name=\"X\", boxpoints=\"all\"))\nfig.update_yaxes(title=\"Axis\")\nfig.update_xaxes(title=\"Pixels\")\nfig.update_layout(title=f\"Box plots of image dimensions based on {N} samples\")\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dim_array = np.array(x_array) * y_array\n\nfig = go.Figure()\nfig.add_trace(go.Box(x=dim_array, name=\"X*Y\", boxpoints=\"all\"))\nfig.update_xaxes(title=\"Pixels\")\nfig.update_layout(title=f\"Box plots of image dimension based on {N} samples\", height=250)\nfig.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Images with biggest dimensions","metadata":{}},{"cell_type":"code","source":"max_x = sample_df.iloc[np.argmax(x_array)]\nmax_y = sample_df.iloc[np.argmax(y_array)]\nmax_dim = sample_df.iloc[np.argmax(dim_array)]\n\ndf = pd.DataFrame()\ndf = df.append(max_x)\ndf = df.append(max_y)\ndf = df.append(max_dim)\nshow_images(df)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Chains with low image or hotel count","metadata":{}},{"cell_type":"markdown","source":"### Chain 18 - 1 hotel with 8 images","metadata":{}},{"cell_type":"code","source":"sample_df = select_data(train_df, 18, None, None)\nshow_images(sample_df, 2, 4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Chain 58 - 2 hotels with total 13 images","metadata":{}},{"cell_type":"code","source":"sample_df = select_data(train_df, 58, None, None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Hotel 4300","metadata":{}},{"cell_type":"code","source":"show_images(sample_df[sample_df[\"hotel_id\"] == 4300], 2, 4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Hotel 56113","metadata":{}},{"cell_type":"code","source":"show_images(sample_df[sample_df[\"hotel_id\"] == 56113], 2, 3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hotels with few images","metadata":{}},{"cell_type":"markdown","source":"### Hotel 14964 - 1 image","metadata":{}},{"cell_type":"code","source":"sample_df = select_data(train_df, None, 14964, None)\nshow_images(sample_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hotel 55404 - 2 images","metadata":{}},{"cell_type":"code","source":"sample_df = select_data(train_df, None, 55404, None)\nshow_images(sample_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test images","metadata":{}},{"cell_type":"code","source":"test_images = os.listdir(data_folder + \"test_images/\")\n\nfig, axs= plt.subplots(1,3, figsize=(22,8))\n\nfor i in range(0, len(test_images)):\n    image = Image.open(data_folder + \"test_images/\" + test_images[i])\n    axs[i].imshow(image)\n    axs[i].set_title(f\"{test_images[i]}\\n{np.shape(image)}\")\n    axs[i].axis(\"off\")       ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}