{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport glob\nfrom tqdm import tqdm\nimport PIL.Image\nfrom joblib import Parallel, delayed\nimport matplotlib.pyplot as plt\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\nimport torchvision\nimport math\nfrom pathlib import Path  \n# from d2l import torch as d2l\n    \n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport cv2\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-16T10:12:43.88973Z","iopub.execute_input":"2022-05-16T10:12:43.890365Z","iopub.status.idle":"2022-05-16T10:12:43.896019Z","shell.execute_reply.started":"2022-05-16T10:12:43.890327Z","shell.execute_reply":"2022-05-16T10:12:43.895362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l_threshold=10\nu_threshold=100\nSIZE = 512","metadata":{"execution":{"iopub.status.busy":"2022-05-15T15:10:07.880182Z","iopub.execute_input":"2022-05-15T15:10:07.88081Z","iopub.status.idle":"2022-05-15T15:10:07.885079Z","shell.execute_reply.started":"2022-05-15T15:10:07.880762Z","shell.execute_reply":"2022-05-15T15:10:07.884292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../working \ndata_folder = \"/kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9/\"\ntrain_folder = os.path.join(data_folder, 'train_images')\nchain_names = os.listdir(train_folder)\n\nprint(os.listdir(data_folder))\nprint(len(chain_names))","metadata":{"execution":{"iopub.status.busy":"2022-05-15T14:52:05.216786Z","iopub.execute_input":"2022-05-15T14:52:05.217572Z","iopub.status.idle":"2022-05-15T14:52:06.001584Z","shell.execute_reply.started":"2022-05-15T14:52:05.21753Z","shell.execute_reply":"2022-05-15T14:52:06.00074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame(columns={'image_id', 'hotel_id'})\nfor hotel_id in tqdm(chain_names):\n    for image_id in os.listdir(os.path.join(train_folder, hotel_id)):\n        train_df = train_df.append({'image_id': image_id, 'hotel_id': hotel_id}, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-15T14:52:06.00327Z","iopub.execute_input":"2022-05-15T14:52:06.004289Z","iopub.status.idle":"2022-05-15T14:53:57.747279Z","shell.execute_reply.started":"2022-05-15T14:52:06.004249Z","shell.execute_reply":"2022-05-15T14:53:57.746392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_filename(path=\"/\"):\n    i=1\n    while True:\n    \n        if str(path)[-i]==\"/\": \n            break\n        i +=1\n\n    #     if i==10: t=False\n    return str(path)[-i+1:]\n# get_filename(path)","metadata":{"execution":{"iopub.status.busy":"2022-05-15T15:07:40.062141Z","iopub.execute_input":"2022-05-15T15:07:40.062467Z","iopub.status.idle":"2022-05-15T15:07:40.074683Z","shell.execute_reply.started":"2022-05-15T15:07:40.062435Z","shell.execute_reply":"2022-05-15T15:07:40.073872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def change_brightness(img, value=50):\n    hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n    h, s, v = cv2.split(hsv)\n    v = cv2.add(v,value)\n    v[v > 255] = 255\n    v[v < 0] = 0\n    final_hsv = cv2.merge((h, s, v))\n    img = cv2.cvtColor(final_hsv, cv2.COLOR_HSV2BGR)\n    return img\n\ndef randomcrop(img, scale=0.5):\n    height, width = int(img.shape[0]*scale), int(img.shape[1]*scale)\n    x = np.random.randint(0, img.shape[1] - int(width))\n    y = np.random.randint(0, img.shape[0] - int(height))\n    cropped = img[y:y+height, x:x+width]\n    resized = cv2.resize(cropped, (img.shape[1], img.shape[0]))    \n    return resized\n\ndef noisy(img, noise_type=\"gauss\"):\n    if noise_type == \"gauss\":\n        image=img.copy() \n        mean=0\n        st=0.7\n        gauss = np.random.normal(mean,st,image.shape)\n        gauss = gauss.astype('uint8')\n        image = cv2.add(image,gauss)\n        return image\n    elif noise_type == \"sp\":\n        image=img.copy() \n        prob = 0.05\n        if len(image.shape) == 2:\n            black = 0\n            white = 255            \n        else:\n            colorspace = image.shape[2]\n            if colorspace == 3:  # RGB\n                black = np.array([0, 0, 0], dtype='uint8')\n                white = np.array([255, 255, 255], dtype='uint8')\n            else:  # RGBA\n                black = np.array([0, 0, 0, 255], dtype='uint8')\n                white = np.array([255, 255, 255, 255], dtype='uint8')\n        probs = np.random.random(image.shape[:2])\n        image[probs < (prob / 2)] = black\n        image[probs > 1 - (prob / 2)] = white\n        return image\n\ndef pad_image(img):\n    w, h, c = np.shape(img)\n    if w > h:\n        pad = int((w - h) / 2)\n        img = cv2.copyMakeBorder(img, 0, 0, pad, pad, cv2.BORDER_CONSTANT, value=0)\n    else:\n        pad = int((h - w) / 2)\n        img = cv2.copyMakeBorder(img, pad, pad, 0, 0, cv2.BORDER_CONSTANT, value=0)\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2022-05-15T14:53:57.748904Z","iopub.execute_input":"2022-05-15T14:53:57.749379Z","iopub.status.idle":"2022-05-15T14:53:57.767628Z","shell.execute_reply.started":"2022-05-15T14:53:57.749333Z","shell.execute_reply":"2022-05-15T14:53:57.766609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_augmentation(folder,train_df, number_of_copies=1,dest_folder=\"../working\",u_threshold=100, hotel_id=0):\n    if dest_folder==\"\": dest_folder=folder\n    for dirname, _, filenames in os.walk(folder):\n        i=0\n        if len(filenames)>u_threshold:\n            filenames = np.random.choice(filenames, size=u_threshold, replace=False)\n            \n        for filename in filenames:             \n            img = cv2.imread(dirname+\"/\"+filename,1)\n            \n            for i in range(number_of_copies):\n                transformed = img\n                \n                if np.random.rand()<0.5:\n                    scale = np.random.rand()/2 + 0.5\n                    transformed = randomcrop(transformed, scale)\n                \n                if np.random.rand()<0.5:\n                    transformed = cv2.flip(transformed, 1)\n\n                if np.random.rand()<0.5:\n                    transformed = cv2.flip(transformed, 0)\n                \n                if np.random.rand()<0.5:\n                    value = np.random.randint(-30, 30) \n                    transformed = change_brightness(transformed, value=value)\n\n                if np.random.rand()<0.3:\n                    size = np.random.randint(3,7)\n                    transformed = cv2.blur(transformed,(size,size))\n\n                if np.random.rand()<0.2:\n                    type_n = np.random.choice(np.array([\"gauss\",\"sp\"]))\n                    transformed = noisy(transformed, type_n)\n                \n                transformed = pad_image(transformed)\n                transformed = cv2.resize(transformed, (SIZE,SIZE), interpolation = cv2.INTER_AREA)\n                # if np.random.rand()>0.25:\n                    \n                name = filename[:-4]+\"_\"+str(i)+\".jpg\"\n                cv2.imwrite(dest_folder+\"/\"+name, transformed)\n                train_df = train_df.append({'image_id': name, 'hotel_id': hotel_id}, ignore_index=True)\n                \n            img = pad_image(img)\n            img = cv2.resize(img, (SIZE,SIZE), interpolation = cv2.INTER_AREA)\n            cv2.imwrite(dest_folder+\"/\" + filename, img)\n    return train_df","metadata":{"execution":{"iopub.status.busy":"2022-05-15T14:53:57.770396Z","iopub.execute_input":"2022-05-15T14:53:57.770649Z","iopub.status.idle":"2022-05-15T14:53:57.788028Z","shell.execute_reply.started":"2022-05-15T14:53:57.770621Z","shell.execute_reply":"2022-05-15T14:53:57.78729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_file(path,train_df,dest_folder=\"../working\"):\n    print(path)\n    img = cv2.imread(path,1)\n    \n    img= cv2.resize(img, (SIZE,SIZE), interpolation = cv2.INTER_AREA)\n    filename = get_filename(path)\n    cv2.imwrite(dest_folder+\"/\" + filename, img)\n    \n    return train_df","metadata":{"execution":{"iopub.status.busy":"2022-05-15T15:09:37.639524Z","iopub.execute_input":"2022-05-15T15:09:37.639985Z","iopub.status.idle":"2022-05-15T15:09:37.646319Z","shell.execute_reply.started":"2022-05-15T15:09:37.639952Z","shell.execute_reply":"2022-05-15T15:09:37.644676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nindexes=np.array(train_df[\"hotel_id\"].value_counts().index)\ncounts = np.array(train_df[\"hotel_id\"].value_counts())\n\n# data_augmentation(os.path.join(train_folder, \"10010\"), number_of_copies=1,dest_folder=\"../working\")\n\n# train_df[train_df['hotel_id'] == \"10010\"].index.values\n# train_df.values[11017][0]\n# path=os.path.join(train_folder, \"10010/\", str(train_df.values[11017][0]))\n# !ls /kaggle/input/hotel-id-to-combat-human-trafficking-2022-fgvc9/train_images/10010/\n\n# img=cv2.imread(path,1)\n\n\n# np.random.choice(train_df[train_df['hotel_id'] == indexes[1]].index.values,10)\n# os.path.join(train_folder, str(indexes[1]),str(counts[20]))\n","metadata":{"execution":{"iopub.status.busy":"2022-05-15T15:09:05.272542Z","iopub.execute_input":"2022-05-15T15:09:05.273266Z","iopub.status.idle":"2022-05-15T15:09:05.303757Z","shell.execute_reply.started":"2022-05-15T15:09:05.273229Z","shell.execute_reply":"2022-05-15T15:09:05.303153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, index in enumerate(indexes):\n    c = counts[i]\n  \n    if c<l_threshold:\n        n_copies = math.floor(l_threshold/c)\n        train_df = data_augmentation(os.path.join(train_folder, str(index)),train_df, number_of_copies=n_copies, dest_folder=\"../working\", hotel_id=index)\n    else:\n        train_df = data_augmentation(os.path.join(train_folder, str(index)),train_df, number_of_copies=0, dest_folder=\"../working\", u_threshold=u_threshold)","metadata":{"execution":{"iopub.status.busy":"2022-05-15T15:14:37.167604Z","iopub.execute_input":"2022-05-15T15:14:37.168347Z","iopub.status.idle":"2022-05-15T15:14:37.220068Z","shell.execute_reply.started":"2022-05-15T15:14:37.168306Z","shell.execute_reply":"2022-05-15T15:14:37.219273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_list=os.listdir(r\"../working\")\nprint(file_list)\nreduced_train_df = train_df[train_df['image_id'].isin(file_list)]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cd /kaggle/working/ & zip -jqr padded_images_256.zip .\n!find . -name \"*.jpg\" -delete\n\n!ls /kaggle/working\n\n# train_df.to_csv('initial_df.csv')\nreduced_train_df.to_csv('final_df.csv')\n\n!ls /kaggle/working\n","metadata":{"execution":{"iopub.status.busy":"2022-05-15T14:53:57.945515Z","iopub.status.idle":"2022-05-15T14:53:57.946308Z","shell.execute_reply.started":"2022-05-15T14:53:57.945992Z","shell.execute_reply":"2022-05-15T14:53:57.94602Z"},"trusted":true},"execution_count":null,"outputs":[]}]}