{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\n\nimport sklearn\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nimport tensorflow.keras as keras\n\nimport multiprocessing\n\nfrom tensorflow.keras.models import load_model \nfrom tensorflow.keras.layers import Dense, Input, Conv2D\nfrom tensorflow.keras.applications import EfficientNetB0, EfficientNetB7\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport tensorflow.keras.layers as L\n\nfrom kaggle_secrets import UserSecretsClient\n\n\nfrom skimage.transform import resize\nimport numpy as np\nimport math\n\nimport wandb","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.71952Z","iopub.execute_input":"2021-07-04T23:36:56.719878Z","iopub.status.idle":"2021-07-04T23:36:56.726709Z","shell.execute_reply.started":"2021-07-04T23:36:56.719844Z","shell.execute_reply":"2021-07-04T23:36:56.725427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cores = multiprocessing.cpu_count()\nprint(f\"CPU Cores: {num_cores}\")","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.728453Z","iopub.execute_input":"2021-07-04T23:36:56.728809Z","iopub.status.idle":"2021-07-04T23:36:56.743478Z","shell.execute_reply.started":"2021-07-04T23:36:56.728775Z","shell.execute_reply":"2021-07-04T23:36:56.742321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/hotel-id-2021-fgvc8/train.csv\")\ntrain = pd.read_csv(\"../input/ssasasa/train2.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.745405Z","iopub.execute_input":"2021-07-04T23:36:56.745706Z","iopub.status.idle":"2021-07-04T23:36:56.872892Z","shell.execute_reply.started":"2021-07-04T23:36:56.745678Z","shell.execute_reply":"2021-07-04T23:36:56.871769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.874601Z","iopub.execute_input":"2021-07-04T23:36:56.874881Z","iopub.status.idle":"2021-07-04T23:36:56.887186Z","shell.execute_reply.started":"2021-07-04T23:36:56.874855Z","shell.execute_reply":"2021-07-04T23:36:56.886238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kaggle_path = \"../input/hotel-id-2021-fgvc8/train_images/\"\ntest_path = \"../input/hotel-id-2021-fgvc8/test_images/\"\ntrain['full_filepath'] = kaggle_path + train.chain.astype(str) +\"/\"+ train.image.astype(str)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.888776Z","iopub.execute_input":"2021-07-04T23:36:56.889084Z","iopub.status.idle":"2021-07-04T23:36:56.905093Z","shell.execute_reply.started":"2021-07-04T23:36:56.889022Z","shell.execute_reply":"2021-07-04T23:36:56.904147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.iloc[0,4]","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.906601Z","iopub.execute_input":"2021-07-04T23:36:56.907111Z","iopub.status.idle":"2021-07-04T23:36:56.920897Z","shell.execute_reply.started":"2021-07-04T23:36:56.907072Z","shell.execute_reply":"2021-07-04T23:36:56.919638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[train.chain.isin([0,1,2])]\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.923109Z","iopub.execute_input":"2021-07-04T23:36:56.923552Z","iopub.status.idle":"2021-07-04T23:36:56.936651Z","shell.execute_reply.started":"2021-07-04T23:36:56.923508Z","shell.execute_reply":"2021-07-04T23:36:56.935792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# n_subsample = 5000\n# train = train.sample(n_subsample)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.939353Z","iopub.execute_input":"2021-07-04T23:36:56.939605Z","iopub.status.idle":"2021-07-04T23:36:56.948131Z","shell.execute_reply.started":"2021-07-04T23:36:56.93958Z","shell.execute_reply":"2021-07-04T23:36:56.947335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"hotel_id\"] = train.hotel_id.astype(\"str\")","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.949546Z","iopub.execute_input":"2021-07-04T23:36:56.9498Z","iopub.status.idle":"2021-07-04T23:36:56.960839Z","shell.execute_reply.started":"2021-07-04T23:36:56.949777Z","shell.execute_reply":"2021-07-04T23:36:56.960025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, = train_test_split(train, test_size=0.3,\n     shuffle = True\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.961802Z","iopub.execute_input":"2021-07-04T23:36:56.962238Z","iopub.status.idle":"2021-07-04T23:36:56.973105Z","shell.execute_reply.started":"2021-07-04T23:36:56.962191Z","shell.execute_reply":"2021-07-04T23:36:56.972344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_val.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.974092Z","iopub.execute_input":"2021-07-04T23:36:56.974462Z","iopub.status.idle":"2021-07-04T23:36:56.988703Z","shell.execute_reply.started":"2021-07-04T23:36:56.974436Z","shell.execute_reply":"2021-07-04T23:36:56.988008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:56.989744Z","iopub.execute_input":"2021-07-04T23:36:56.99014Z","iopub.status.idle":"2021-07-04T23:36:57.008169Z","shell.execute_reply.started":"2021-07-04T23:36:56.990112Z","shell.execute_reply":"2021-07-04T23:36:57.007162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_classes = X_train.hotel_id.nunique()\n\nBATCH_SIZE = 4\nSTEPS_PER_EPOCH = len(X_train) // BATCH_SIZE\nEPOCHS = 15\n\nIMG_HEIGHT = 226\nIMG_WIDTH = 226\nIMG_SIZE = (IMG_HEIGHT, IMG_WIDTH)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:57.009513Z","iopub.execute_input":"2021-07-04T23:36:57.009814Z","iopub.status.idle":"2021-07-04T23:36:57.023259Z","shell.execute_reply.started":"2021-07-04T23:36:57.009784Z","shell.execute_reply":"2021-07-04T23:36:57.022269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gen = ImageDataGenerator(rescale=1./255, validation_split=0.2)\n\ntrain_gen = gen.flow_from_dataframe(\n    X_train,\n#     directory=\"../input/hotel-id-2021-fgvc8/train_images\",\n    x_col=\"full_filepath\",\n    y_col=\"hotel_id\",\n    weight_col=None,\n    target_size=(IMG_HEIGHT, IMG_WIDTH),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    subset=\"training\",\n    interpolation=\"nearest\",\n    validate_filenames=False)\n    \nval_gen = gen.flow_from_dataframe(\n    X_val,\n#     directory=\"../input/hotel-id-2021-fgvc8/train_images\",\n    x_col=\"full_filepath\",\n    y_col=\"hotel_id\",\n    weight_col=None,\n    target_size=(IMG_HEIGHT, IMG_WIDTH),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    subset=\"validation\",\n    interpolation=\"nearest\",\n    validate_filenames=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:57.0244Z","iopub.execute_input":"2021-07-04T23:36:57.024648Z","iopub.status.idle":"2021-07-04T23:36:57.051441Z","shell.execute_reply.started":"2021-07-04T23:36:57.024625Z","shell.execute_reply":"2021-07-04T23:36:57.050522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Based on https://www.tensorflow.org/api_docs/python/tf/keras/utils/Sequence\n# https://github.com/keras-team/keras/issues/12847\n# https://stanford.edu/~shervine/blog/keras-how-to-generate-data-on-the-fly\n# https://keunwoochoi.wordpress.com/2017/08/24/tip-fit_generator-in-keras-how-to-parallelise-correctly/\n\nclass HotelBatchSequence(tf.keras.utils.Sequence):\n    \n    def __init__(self, x_set, y_set, batch_size,\n                 img_size = (IMG_HEIGHT, IMG_WIDTH),\n                 augment = False):\n        \"\"\"\n        `x_set` is list of paths to the images\n        `y_set` are the associated classes.\n\n        \"\"\"\n        \n        self.x = x_set\n        self.y = y_set\n        self.batch_size = batch_size\n        self.img_size = img_size\n    \n    def __len__(self):\n        \"\"\"Denotes the number of batches per epoch\"\"\"\n        return math.ceil(len(self.x) / self.batch_size)\n    \n    def __getitem__(self, idx):\n        \"\"\"Generate one batch of data\"\"\"\n        \n        first_id = idx * self.batch_size\n        last_id =  (idx + 1) * (self.batch_size)\n        \n        batch_x = self.x[first_id:last_id]\n        batch_y = self.y[first_id:last_id]\n        \n        #Xs = np.array([resize(imread(file_name), self.img_size)\n        #      for file_name in batch_x])\n        # \n        #ys = np.array(batch_y)\n        \n        output = np.array([\n            resize(cv2.imread(file_name), self.img_size)\n                   for file_name in batch_x]), np.array(batch_y)\n        \n        return output","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:57.052646Z","iopub.execute_input":"2021-07-04T23:36:57.053059Z","iopub.status.idle":"2021-07-04T23:36:57.059662Z","shell.execute_reply.started":"2021-07-04T23:36:57.053005Z","shell.execute_reply":"2021-07-04T23:36:57.058729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" efficientnet = EfficientNetB7(include_top=True, \n                               weights=None, \n                               input_shape = (IMG_HEIGHT, IMG_WIDTH, 3),\n                               classes = n_classes\n)\n# model = tf.keras.Sequential([\n#    EfficientNetB7(\n#        input_shape=(IMG_WIDTH, IMG_HEIGHT, 3),\n#        weights=None,\n#        include_top=False,\n#        classes = n_classes\n#     ),\n#     L.GlobalAveragePooling2D(),\n#      L.Dense(n_classes, activation='softmax')\n# ])\n\n#model = load_model('../input/find-room/tfmodels/weights.15.hdf5')\n\n# model.load_weights('../input/find-room/tfmodels/weights.15.hdf5')\n# efficientnet.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:36:57.060746Z","iopub.execute_input":"2021-07-04T23:36:57.061008Z","iopub.status.idle":"2021-07-04T23:37:02.786601Z","shell.execute_reply.started":"2021-07-04T23:36:57.060984Z","shell.execute_reply":"2021-07-04T23:37:02.785584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" model = efficientnet\n\nmodel.compile(optimizer = 'SGD',\n              loss = 'categorical_crossentropy',\n              metrics = 'accuracy')","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:37:02.788201Z","iopub.execute_input":"2021-07-04T23:37:02.788597Z","iopub.status.idle":"2021-07-04T23:37:02.813932Z","shell.execute_reply.started":"2021-07-04T23:37:02.788552Z","shell.execute_reply":"2021-07-04T23:37:02.812613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Source: https://gist.github.com/Callidior/747eb767862c9d48f9d900a6373b16d1\n# Author: Callidior\n\n# Also: https://gist.github.com/jeremyjordan/5a222e04bb78c242f5763ad40626c452\n\nclass SGDR(tf.keras.callbacks.Callback):\n    \"\"\"\n    \n    # Source: https://gist.github.com/Callidior/747eb767862c9d48f9d900a6373b16d1\n    # Author: Callidior\n\n    This callback implements the learning rate schedule for\n    Stochastic Gradient Descent with warm Restarts (SGDR),\n    as proposed by Loshchilov & Hutter (https://arxiv.org/abs/1608.03983).\n    \n    The learning rate at each epoch is computed as:\n    lr(i) = min_lr + 0.5 * (max_lr - min_lr) * (1 + cos(pi * i/num_epochs))\n    \n    Here, num_epochs is the number of epochs in the current cycle, which starts\n    with base_epochs initially and is multiplied by mul_epochs after each cycle.\n    \n    # Example\n        ```python\n            sgdr = CyclicLR(min_lr=0.0, max_lr=0.05,\n                                base_epochs=10, mul_epochs=2)\n            model.compile(optimizer=keras.optimizers.SGD(decay=1e-4, momentum=0.9),\n                          loss=loss)\n            model.fit(X_train, Y_train, callbacks=[sgdr])\n        ```\n    \n    # Arguments\n        min_lr: minimum learning rate reached at the end of each cycle.\n        max_lr: maximum learning rate used at the beginning of each cycle.\n        base_epochs: number of epochs in the first cycle.\n        mul_epochs: factor with which the number of epochs is multiplied\n                after each cycle.\n    \"\"\"\n\n    def __init__(self, min_lr=0.0, max_lr=0.05, base_epochs=10, mul_epochs=2):\n        super(SGDR, self).__init__()\n\n        self.min_lr = min_lr\n        self.max_lr = max_lr\n        self.base_epochs = base_epochs\n        self.mul_epochs = mul_epochs\n\n        self.cycles = 0.\n        self.cycle_iterations = 0.\n        self.trn_iterations = 0.\n\n        self._reset()\n\n    def _reset(self, new_min_lr=None, new_max_lr=None,\n               new_base_epochs=None, new_mul_epochs=None):\n        \"\"\"Resets cycle iterations.\"\"\"\n        \n        if new_min_lr != None:\n            self.min_lr = new_min_lr\n        if new_max_lr != None:\n            self.max_lr = new_max_lr\n        if new_base_epochs != None:\n            self.base_epochs = new_base_epochs\n        if new_mul_epochs != None:\n            self.mul_epochs = new_mul_epochs\n        self.cycles = 0.\n        self.cycle_iterations = 0.\n        \n    def sgdr(self):\n        \n        cycle_epochs = self.base_epochs * (self.mul_epochs ** self.cycles)\n        return self.min_lr + 0.5 * (self.max_lr - self.min_lr) * (1 + np.cos(np.pi * (self.cycle_iterations + 1) / cycle_epochs))\n        \n    def on_train_begin(self, logs=None):\n        \n        if self.cycle_iterations == 0:\n            K.set_value(self.model.optimizer.lr, self.max_lr)\n        else:\n            K.set_value(self.model.optimizer.lr, self.sgdr())\n            \n    def on_epoch_end(self, epoch, logs=None):\n        \n        logs = logs or {}\n        logs['lr'] = K.get_value(self.model.optimizer.lr)\n        \n        self.trn_iterations += 1\n        self.cycle_iterations += 1\n        if self.cycle_iterations >= self.base_epochs * (self.mul_epochs ** self.cycles):\n            self.cycles += 1\n            self.cycle_iterations = 0\n            K.set_value(self.model.optimizer.lr, self.max_lr)\n        else:\n            K.set_value(self.model.optimizer.lr, self.sgdr())","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:37:02.815427Z","iopub.execute_input":"2021-07-04T23:37:02.815741Z","iopub.status.idle":"2021-07-04T23:37:02.827483Z","shell.execute_reply.started":"2021-07-04T23:37:02.815713Z","shell.execute_reply":"2021-07-04T23:37:02.826525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.init()\nwandb_callback = wandb.keras.WandbCallback(log_weights=True)\nmodel_checkpoint = tf.keras.callbacks.ModelCheckpoint(\"/kaggle/working/weights.{epoch:02d}.hdf5\")\ncosine_annealing_lr = SGDR(min_lr=0.0, max_lr=0.05, base_epochs=EPOCHS, mul_epochs=2)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:37:02.828433Z","iopub.execute_input":"2021-07-04T23:37:02.828803Z","iopub.status.idle":"2021-07-04T23:37:09.828658Z","shell.execute_reply.started":"2021-07-04T23:37:02.828777Z","shell.execute_reply":"2021-07-04T23:37:09.827455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_gen,\n                    steps_per_epoch = STEPS_PER_EPOCH,\n                    workers = num_cores,\n                    epochs = 10,\n                    max_queue_size = 10,\n                    callbacks=[\n                         wandb_callback, \n                         model_checkpoint,\n#                         cosine_annealing_lr\n                    ],\n                   verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T23:37:09.830951Z","iopub.execute_input":"2021-07-04T23:37:09.831385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/hotel-id-2021-fgvc8/sample_submission.csv\")\n# sub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = ImageDataGenerator().flow_from_dataframe(\n    sub,\n    directory=test_path,\n    x_col=\"image\",\n    target_size=(IMG_HEIGHT, IMG_WIDTH),\n    batch_size=BATCH_SIZE,\n    classes=None,\n    class_mode=None,\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_gen)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[\"hotel_id\"] = [\" \".join(np.argwhere(i > 0.3).astype(\"str\").flatten()) for i in pred]\nsub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}