{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Intro\nNotebook to download [Hotels-50K dataset](https://github.com/GWUvision/Hotels-50K) images based on the [download_train.py](https://github.com/GWUvision/Hotels-50K/blob/master/download_train.py) script.\n\n### Disclaimer: We are currently not sure if using external datasets is allowed in this competition!\nYou can see [Use of External Data Sets](https://www.kaggle.com/competitions/hotel-id-to-combat-human-trafficking-2022-fgvc9/discussion/317922) discussion for details. The competition host did not confirm whether we can use external data yet.","metadata":{}},{"cell_type":"markdown","source":"# Download dataset with image info from github repo","metadata":{}},{"cell_type":"code","source":"!mkdir hotels-50k\n!wget -P hotels-50k https://github.com/GWUvision/Hotels-50K/raw/master/input/dataset.tar.gz\n!tar -xvzf hotels-50k/dataset.tar.gz -C hotels-50k\n!rm hotels-50k/dataset.tar.gz","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:32.415631Z","iopub.execute_input":"2022-04-17T15:12:32.416193Z","iopub.status.idle":"2022-04-17T15:12:38.044825Z","shell.execute_reply.started":"2022-04-17T15:12:32.416097Z","shell.execute_reply":"2022-04-17T15:12:38.043585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data info","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:38.047233Z","iopub.execute_input":"2022-04-17T15:12:38.047501Z","iopub.status.idle":"2022-04-17T15:12:38.051868Z","shell.execute_reply.started":"2022-04-17T15:12:38.047465Z","shell.execute_reply":"2022-04-17T15:12:38.051024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chain_df = pd.read_csv(\"./hotels-50k/dataset/chain_info.csv\")\ndisplay(chain_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:38.053236Z","iopub.execute_input":"2022-04-17T15:12:38.053547Z","iopub.status.idle":"2022-04-17T15:12:38.084610Z","shell.execute_reply.started":"2022-04-17T15:12:38.053508Z","shell.execute_reply":"2022-04-17T15:12:38.084031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hotel_df = pd.read_csv(\"./hotels-50k/dataset/hotel_info.csv\")\ndisplay(hotel_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:38.086257Z","iopub.execute_input":"2022-04-17T15:12:38.086596Z","iopub.status.idle":"2022-04-17T15:12:38.162884Z","shell.execute_reply.started":"2022-04-17T15:12:38.086568Z","shell.execute_reply":"2022-04-17T15:12:38.162091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('./hotels-50k/dataset/train_set.csv', header=None, \n                       names=['image_id', 'hotel_id', 'url', 'source', 'timestamp'])\n\ndisplay(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:38.164112Z","iopub.execute_input":"2022-04-17T15:12:38.164345Z","iopub.status.idle":"2022-04-17T15:12:40.571862Z","shell.execute_reply.started":"2022-04-17T15:12:38.164316Z","shell.execute_reply":"2022-04-17T15:12:40.571052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check Hotels-50k data","metadata":{}},{"cell_type":"code","source":"import plotly\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:40.573168Z","iopub.execute_input":"2022-04-17T15:12:40.573404Z","iopub.status.idle":"2022-04-17T15:12:42.113693Z","shell.execute_reply.started":"2022-04-17T15:12:40.573374Z","shell.execute_reply":"2022-04-17T15:12:42.112963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df = train_df.merge(hotel_df, on=\"hotel_id\").merge(chain_df, on=\"chain_id\")\ndata_df[\"image_id\"] = data_df[\"image_id\"].astype(str)\ndata_df[\"hotel_id\"] = data_df[\"hotel_id\"].astype(str)\ndata_df[\"chain_id\"] = data_df[\"chain_id\"].astype(str)\n\ndisplay(data_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:42.114809Z","iopub.execute_input":"2022-04-17T15:12:42.115049Z","iopub.status.idle":"2022-04-17T15:12:46.508861Z","shell.execute_reply.started":"2022-04-17T15:12:42.115021Z","shell.execute_reply":"2022-04-17T15:12:46.508051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset size","metadata":{}},{"cell_type":"code","source":"print(\"Image count:\", len(data_df))\nprint(\"Hotel count:\", len(data_df[\"hotel_id\"].unique()))\nprint(\"Chain count:\", len(data_df[\"chain_id\"].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:46.510324Z","iopub.execute_input":"2022-04-17T15:12:46.510787Z","iopub.status.idle":"2022-04-17T15:12:46.688447Z","shell.execute_reply.started":"2022-04-17T15:12:46.510732Z","shell.execute_reply":"2022-04-17T15:12:46.686719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hotel and image count per chain","metadata":{}},{"cell_type":"code","source":"chain_group_df = data_df.groupby([\"chain_name\"]).agg({\"hotel_id\": [pd.Series.nunique], \"image_id\" : [pd.Series.nunique]})\nchain_group_df.columns = [\"_\".join(x) for x in chain_group_df.columns.ravel()]\nchain_group_df = chain_group_df.reset_index().sort_values(\"hotel_id_nunique\")[::-1]","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:46.689671Z","iopub.execute_input":"2022-04-17T15:12:46.689882Z","iopub.status.idle":"2022-04-17T15:12:47.287000Z","shell.execute_reply.started":"2022-04-17T15:12:46.689856Z","shell.execute_reply":"2022-04-17T15:12:47.286180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(chain_group_df, x=\"chain_name\", y=\"hotel_id_nunique\",\n                 size=\"image_id_nunique\", color = \"image_id_nunique\",\n                 hover_name = None,\n                 log_y=True, size_max=75)\n\nfig.update_yaxes(title_text=\"Hotel count\")\nfig.update_xaxes(title_text=\"Chain ID\")\nfig.update_layout(title=\"Hotel and image count per chain\", coloraxis=dict(colorbar=dict(title=\"Image count\")))\nfig.update_traces(hovertemplate=\"Chain: %{x} <br>Hotel count: %{y:%d}<br>Image count: %{marker.size:%d}\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:47.289604Z","iopub.execute_input":"2022-04-17T15:12:47.289826Z","iopub.status.idle":"2022-04-17T15:12:48.386610Z","shell.execute_reply.started":"2022-04-17T15:12:47.289799Z","shell.execute_reply":"2022-04-17T15:12:48.385748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image count per hotel","metadata":{}},{"cell_type":"code","source":"group_df = data_df.groupby([\"hotel_id\"]).size().to_frame(\"image_count\").sort_values(\"image_count\")[::-1].reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:48.388002Z","iopub.execute_input":"2022-04-17T15:12:48.388317Z","iopub.status.idle":"2022-04-17T15:12:48.609052Z","shell.execute_reply.started":"2022-04-17T15:12:48.388275Z","shell.execute_reply":"2022-04-17T15:12:48.608130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(group_df, x=\"image_count\", nbins=100, marginal=\"box\", height=500)\nfig.update_layout(title=\"Distribution of image count per hotel\")\nfig.update_traces(hovertemplate=\"Image count: %{x} <br>Hotel count: %{y:%d}\")\nfig.update_yaxes(title_text=\"Hotel count\")\nfig.update_xaxes(title_text=\"Image count\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:48.610599Z","iopub.execute_input":"2022-04-17T15:12:48.610848Z","iopub.status.idle":"2022-04-17T15:12:48.953357Z","shell.execute_reply.started":"2022-04-17T15:12:48.610817Z","shell.execute_reply":"2022-04-17T15:12:48.952624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image count per source\nImages come from two different sources: travel_website and traffickcam","metadata":{}},{"cell_type":"code","source":"group_df = data_df.groupby([\"source\"]).size().to_frame(\"image_count\").sort_values(\"image_count\")[::-1].reset_index()\n\nfig = px.bar(group_df, x=\"source\", y=\"image_count\", height=500)\nfig.update_layout(title=\"Image count per source\")\nfig.update_traces(hovertemplate=\"Source: %{x:%d} <br>Image count: %{y:%d}\")\nfig.update_yaxes(title_text=\"Image count\")\nfig.update_xaxes(title_text=\"Source\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:16:11.520421Z","iopub.execute_input":"2022-04-17T15:16:11.520749Z","iopub.status.idle":"2022-04-17T15:16:11.689745Z","shell.execute_reply.started":"2022-04-17T15:16:11.520713Z","shell.execute_reply":"2022-04-17T15:16:11.689162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample 50 hotels with more than 10 and less than 100 images","metadata":{}},{"cell_type":"code","source":"hotel_group_df = data_df.groupby(by=[\"hotel_id\"])[\"image_id\"].count().to_frame(\"image_count\")","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.136941Z","iopub.status.idle":"2022-04-17T15:12:49.137761Z","shell.execute_reply.started":"2022-04-17T15:12:49.137462Z","shell.execute_reply":"2022-04-17T15:12:49.137492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_hotels = hotel_group_df[(hotel_group_df[\"image_count\"] > 10) & (hotel_group_df[\"image_count\"] < 100)]\nprint(\"Number of hotels with more than 10 images and less than 100:\", len(sample_hotels))\nsample_hotels = sample_hotels.sample(50, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.139460Z","iopub.status.idle":"2022-04-17T15:12:49.140409Z","shell.execute_reply.started":"2022-04-17T15:12:49.140111Z","shell.execute_reply":"2022-04-17T15:12:49.140142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = data_df[data_df[\"hotel_id\"].isin(sample_hotels.index)].reset_index(drop=True)\nprint(\"Sampled images:\", len(sample_df))","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.141748Z","iopub.status.idle":"2022-04-17T15:12:49.142142Z","shell.execute_reply.started":"2022-04-17T15:12:49.141936Z","shell.execute_reply":"2022-04-17T15:12:49.141955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chain_group_df = sample_df.groupby([\"chain_name\"]).agg({\"hotel_id\": [pd.Series.nunique], \"image_id\" : [pd.Series.nunique]})\nchain_group_df.columns = [\"_\".join(x) for x in chain_group_df.columns.ravel()]\nchain_group_df = chain_group_df.reset_index().sort_values(\"hotel_id_nunique\")[::-1]\n\nfig = px.scatter(chain_group_df, x=\"chain_name\", y=\"hotel_id_nunique\",\n                 size=\"image_id_nunique\", color = \"image_id_nunique\",\n                 hover_name = None,\n                 size_max=75)\n\nfig.update_yaxes(title_text=\"Hotel count\")\nfig.update_xaxes(title_text=\"Chain ID\")\nfig.update_layout(title=\"Sampled data <br>Hotel and image count per chain\", coloraxis=dict(colorbar=dict(title=\"Image count\")))\nfig.update_traces(hovertemplate=\"Chain: %{x} <br>Hotel count: %{y:%d}<br>Image count: %{marker.size:%d}\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.143778Z","iopub.status.idle":"2022-04-17T15:12:49.144156Z","shell.execute_reply.started":"2022-04-17T15:12:49.143951Z","shell.execute_reply":"2022-04-17T15:12:49.143968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Download sampled images","metadata":{}},{"cell_type":"markdown","source":"## Prepare to download images\nThe SSL certificate of the image urls is expired so we have to handle it.","metadata":{}},{"cell_type":"code","source":"from __future__ import print_function\nimport csv, multiprocessing, cv2, os\nimport numpy as np\nimport urllib\nimport urllib.request\n\nimport ssl\n\nctx = ssl.create_default_context()\nctx.check_hostname = False\nctx.verify_mode = ssl.CERT_NONE","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.145047Z","iopub.status.idle":"2022-04-17T15:12:49.145345Z","shell.execute_reply.started":"2022-04-17T15:12:49.145187Z","shell.execute_reply":"2022-04-17T15:12:49.145203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_folder = \"hotels-50k/images\"\noutput_image_folder  = output_folder + \"/train\"\n\nos.makedirs(output_image_folder)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.147072Z","iopub.status.idle":"2022-04-17T15:12:49.147636Z","shell.execute_reply.started":"2022-04-17T15:12:49.147443Z","shell.execute_reply":"2022-04-17T15:12:49.147464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Download images","metadata":{}},{"cell_type":"markdown","source":"We will download the images without padding or resizing and we will keep the original folder structure: hotels-50k/images/train/chain_id/hotel_id/source/image_id.jpeg","metadata":{}},{"cell_type":"code","source":"def url_to_image(url):\n    resp = urllib.request.urlopen(url, context=ctx)\n    image = np.asarray(bytearray(resp.read()), dtype=\"uint8\")\n    image = cv2.imdecode(image, cv2.IMREAD_UNCHANGED)\n    return image\n\n\ndef download_images(imList):\n    # 2d list, rows are samples\n    # columns: \"chain_id\", \"hotel_id\", \"source\", \"image_id\", \"url\"\n    for im in imList:\n        try:\n            saveDir = os.path.join(output_image_folder, im[0], im[1], im[2])\n            if not os.path.exists(saveDir):\n                os.makedirs(saveDir)\n\n            savePath = os.path.join(saveDir, str(im[3])+'.'+im[4].split('.')[-1])\n\n            if not os.path.isfile(savePath):\n                img = url_to_image(im[4])\n                cv2.imwrite(savePath,img)\n            else:\n                print('Already exists: ' + savePath)\n        except Exception as e:\n            print(e, ': ' + im[4])","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.148548Z","iopub.status.idle":"2022-04-17T15:12:49.148844Z","shell.execute_reply.started":"2022-04-17T15:12:49.148689Z","shell.execute_reply":"2022-04-17T15:12:49.148705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nimage_data_array = sample_df[[\"chain_id\", \"hotel_id\", \"source\", \"image_id\", \"url\"]].values\n\npool = multiprocessing.Pool()\nNUM_THREADS = multiprocessing.cpu_count()\nfor cpu in range(NUM_THREADS):\n    pool.apply_async(download_images,[image_data_array[cpu::NUM_THREADS]])\n\npool.close()\npool.join()","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.150340Z","iopub.status.idle":"2022-04-17T15:12:49.150900Z","shell.execute_reply.started":"2022-04-17T15:12:49.150711Z","shell.execute_reply":"2022-04-17T15:12:49.150731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Not every image is available, lets check how many images were successfully downloaded","metadata":{}},{"cell_type":"code","source":"!find {output_image_folder} -type f | wc -l","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.151916Z","iopub.status.idle":"2022-04-17T15:12:49.152474Z","shell.execute_reply.started":"2022-04-17T15:12:49.152281Z","shell.execute_reply":"2022-04-17T15:12:49.152301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check downloaded data","metadata":{}},{"cell_type":"code","source":"# update the sample data frame with path, image name and whether it was downloaded\nsample_df[\"downloaded\"] = False\nsample_df[\"image_name\"] = None\nsample_df[\"image_folder\"] = None\n\nfor index, row in sample_df.iterrows():\n    image_folder = os.path.join(output_image_folder, row[\"chain_id\"], row[\"hotel_id\"], row[\"source\"])\n    image_name   = row[\"image_id\"] + '.'+ row[\"url\"].split('.')[-1]\n    image_path   = os.path.join(image_folder, image_name)\n    if os.path.exists(image_path):\n        sample_df.loc[index, \"downloaded\"] = True\n        sample_df.loc[index, \"image_name\"] = image_name\n        sample_df.loc[index, \"image_folder\"] = image_folder","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.153424Z","iopub.status.idle":"2022-04-17T15:12:49.153949Z","shell.execute_reply.started":"2022-04-17T15:12:49.153760Z","shell.execute_reply":"2022-04-17T15:12:49.153779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(sample_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.155015Z","iopub.status.idle":"2022-04-17T15:12:49.155502Z","shell.execute_reply.started":"2022-04-17T15:12:49.155313Z","shell.execute_reply":"2022-04-17T15:12:49.155333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of downloaded images should be the same as number of images in the output_image_folder\nprint(\"Number of downloaded images:\", sample_df[\"downloaded\"].sum())","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.156353Z","iopub.status.idle":"2022-04-17T15:12:49.157366Z","shell.execute_reply.started":"2022-04-17T15:12:49.157154Z","shell.execute_reply":"2022-04-17T15:12:49.157178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save sample df to csv\nsample_df.to_csv(\"hotels-50k/sample.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.158392Z","iopub.status.idle":"2022-04-17T15:12:49.158929Z","shell.execute_reply.started":"2022-04-17T15:12:49.158757Z","shell.execute_reply":"2022-04-17T15:12:49.158774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display some images","metadata":{}},{"cell_type":"code","source":"from matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.160077Z","iopub.status.idle":"2022-04-17T15:12:49.160601Z","shell.execute_reply.started":"2022-04-17T15:12:49.160439Z","shell.execute_reply":"2022-04-17T15:12:49.160456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"downloaded_df = sample_df[sample_df[\"downloaded\"]].sample(6, random_state=42)\n\nfig, axes = plt.subplots(3,2, figsize=(16,20))\naxes = axes.ravel()\n\nfor i in range(6):\n    sample = downloaded_df.iloc[i]\n    img = cv2.imread(os.path.join(sample[\"image_folder\"], sample[\"image_name\"])) \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    axes[i].imshow(img)\n    axes[i].set_title(f\"Chain: {sample.chain_name}\\n\" + \n                     f\"Hotel_name: {sample.hotel_name}\\n\"\n                     f\"Image: {sample.image_id}\\n\" +\n                     f\"Time: {sample.timestamp}\\n\" + \n                     f\"Size: {np.shape(img)}\")","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.161557Z","iopub.status.idle":"2022-04-17T15:12:49.162166Z","shell.execute_reply.started":"2022-04-17T15:12:49.161962Z","shell.execute_reply":"2022-04-17T15:12:49.162001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create zip and clean up\nCompress the downloaded images to zip file and delete the data.","metadata":{}},{"cell_type":"code","source":"!zip -r -qq hotels-50K-sample.zip hotels-50k\n!rm -rf hotels-50k","metadata":{"execution":{"iopub.status.busy":"2022-04-17T15:12:49.163513Z","iopub.status.idle":"2022-04-17T15:12:49.163966Z","shell.execute_reply.started":"2022-04-17T15:12:49.163793Z","shell.execute_reply":"2022-04-17T15:12:49.163810Z"},"trusted":true},"execution_count":null,"outputs":[]}]}