{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pathlib\n\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image as pil_img\nimport cv2\n\nfrom tqdm import tqdm\nfrom joblib import Parallel, delayed","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-21T14:44:27.21244Z","iopub.execute_input":"2022-04-21T14:44:27.212786Z","iopub.status.idle":"2022-04-21T14:44:27.618269Z","shell.execute_reply.started":"2022-04-21T14:44:27.2127Z","shell.execute_reply":"2022-04-21T14:44:27.617427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Globals","metadata":{}},{"cell_type":"code","source":"# directory of the data\ndata_dir = pathlib.Path(\"/kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9/\")\n\n# Work directory, where to store the data\nworking_dir = pathlib.Path(\"./\")\n\n# Locations of the train, mask and test images in the data directory\ntrain_dir = data_dir / pathlib.Path(\"train_images\")\ntrain_mask_dir = data_dir / pathlib.Path(\"train_masks\")\ntest_dir = data_dir / pathlib.Path(\"test_images\")\n\n# Where to store the data in the work directory\ntrain_out_dir = working_dir / pathlib.Path(\"pad_and_resize/train_images\")\ntest_out_dir = working_dir / pathlib.Path(\"pad_and_resize/test_images\")\n\ntrain_out_dir.mkdir(parents=True, exist_ok=True)\ntest_out_dir.mkdir(parents=True, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T14:44:27.619617Z","iopub.execute_input":"2022-04-21T14:44:27.619841Z","iopub.status.idle":"2022-04-21T14:44:27.62588Z","shell.execute_reply.started":"2022-04-21T14:44:27.619816Z","shell.execute_reply":"2022-04-21T14:44:27.625353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 42\n\n# Wheter to PAD the images\nPAD = True\n\n# The size to resize the images to\nPATCH = (512, 512)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T14:44:27.626987Z","iopub.execute_input":"2022-04-21T14:44:27.627388Z","iopub.status.idle":"2022-04-21T14:44:27.638635Z","shell.execute_reply.started":"2022-04-21T14:44:27.627358Z","shell.execute_reply":"2022-04-21T14:44:27.637734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"chain_names = os.listdir(train_dir)\n\ntrain_file = working_dir / pathlib.Path(\"train.csv\")\nmask_file = working_dir / pathlib.Path(\"mask.csv\")\ntest_file = working_dir / pathlib.Path(\"test.csv\")\n\nif not train_file.exists():\n    train_df = pd.DataFrame(columns={'image_id', 'hotel_id'})\n    for hotel_id in chain_names:\n        for image_id in os.listdir(train_dir / hotel_id):\n            train_df = train_df.append({'image_id': image_id, 'hotel_id': hotel_id}, ignore_index=True)\n    train_df.to_csv(train_file, index=False)\nelse:\n    train_df = pd.read_csv(train_file)\n\nif not mask_file.exists():\n    mask_df = pd.DataFrame(columns={'image_id'})\n    for image_id in os.listdir(train_mask_dir):\n        mask_df = mask_df.append({'image_id': image_id}, ignore_index=True)\n    mask_df.to_csv(mask_file, index=False)\nelse:\n    mask_df = pd.read_csv(mask_file)    \n\nif not test_file.exists():\n    test_df = pd.DataFrame(columns={'image_id'})\n    for image_id in os.listdir(test_dir):\n        test_df = test_df.append({'image_id': image_id}, ignore_index=True)\n    test_df.to_csv(test_file, index=False)\nelse:\n    test_df = pd.read_csv(test_file)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T14:44:27.640022Z","iopub.execute_input":"2022-04-21T14:44:27.640547Z","iopub.status.idle":"2022-04-21T14:44:29.141827Z","shell.execute_reply.started":"2022-04-21T14:44:27.64051Z","shell.execute_reply":"2022-04-21T14:44:29.14055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of train images:\", len(train_df))\nprint(\"Number of test images:\", len(test_df))\nprint(\"Number of different classes:\", len(chain_names))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T14:44:29.142708Z","iopub.status.idle":"2022-04-21T14:44:29.143428Z","shell.execute_reply.started":"2022-04-21T14:44:29.14319Z","shell.execute_reply":"2022-04-21T14:44:29.143217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessor","metadata":{}},{"cell_type":"code","source":"def process_img(img_path: pathlib.Path):\n    img = cv2.imread(str(img_path))\n    \n    if PAD: img = pad(img)\n    \n    return cv2.resize(img, PATCH)\n\ndef save_img(img_path: pathlib.Path, img: np.array):\n    cv2.imwrite(str(img_path), img)\n\n\"\"\"\ndef pad(img):\n    w, h, c = np.shape(img)\n    const = 0\n        \n    if w == h: return img\n    elif (w - h) % 2 != 0: const = 1\n        \n    if w < h:\n        half_py = (h - w) // 2       \n        return cv2.copyMakeBorder(img, 0, 0, half_py, half_py + const, cv2.BORDER_CONSTANT, value=0)\n    elif h < w:\n        half_px = (w - h) // 2\n        return cv2.copyMakeBorder(img, half_px, half_px + const, 0, 0, cv2.BORDER_CONSTANT, value=0)\n\"\"\"    \n\ndef pad(img):\n    w, h, c = np.shape(img)\n    if w > h:\n        pad = int((w - h) / 2)\n        img = cv2.copyMakeBorder(img, 0, 0, pad, pad, cv2.BORDER_CONSTANT, value=0)\n    else:\n        pad = int((h - w) / 2)\n        img = cv2.copyMakeBorder(img, pad, pad, 0, 0, cv2.BORDER_CONSTANT, value=0)\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2022-04-21T14:44:29.144913Z","iopub.status.idle":"2022-04-21T14:44:29.145453Z","shell.execute_reply.started":"2022-04-21T14:44:29.145289Z","shell.execute_reply":"2022-04-21T14:44:29.145308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_train(data_folder, chain_name, out_dir):\n    chain_folder = data_folder / chain_name\n    \n    for img_name in os.listdir(chain_folder):\n        img = process_img(chain_folder / img_name)\n        save_img(out_dir / img_name, img)\n\n\ndfs_proc = Parallel(n_jobs=4, prefer='threads')(delayed(process_train)(train_dir, chain_names[i], train_out_dir) for i in range(0, len(chain_names)))","metadata":{"execution":{"iopub.status.busy":"2022-04-21T14:44:29.146334Z","iopub.status.idle":"2022-04-21T14:44:29.146998Z","shell.execute_reply.started":"2022-04-21T14:44:29.146746Z","shell.execute_reply":"2022-04-21T14:44:29.146775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cd /kaggle/working/pad_and_resize & zip -jqr images.zip .\n!find . -name \"*.jpg\" -delete","metadata":{},"execution_count":null,"outputs":[]}]}