{"cells":[{"metadata":{},"cell_type":"markdown","source":"","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Import Packages**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# !pip install ipython-autotime\n# %load_ext autotime\n\nimport numpy as np\nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n%matplotlib inline\nimport plotly.express as px\nimport plotly.figure_factory as ff\nimport plotly.graph_objects as go\nfrom scipy import stats\nimport cv2\nimport glob\nfrom keras.preprocessing.image import ImageDataGenerator\n# from keras.applications import MobileNetV2\nfrom keras.utils import to_categorical\nfrom keras.layers import Dense\nfrom keras import Model\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom keras.models import load_model, Model\nfrom keras.applications import MobileNetV2\nfrom keras.optimizers import Adam\n# from tensorflow.keras.applications.xception import Xception\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nfrom collections import Counter\n\n\n# import efficientnet.tfkeras as efn","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df=pd.read_csv('../input/landmark-recognition-2020/train.csv')\nprint(train_df.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"landmark_counts=pd.value_counts(train_df[\"landmark_id\"])\nlandmark_counts=landmark_counts.reset_index()\nlandmark_counts.rename(columns={\"index\":'landmark_ids','landmark_id':'count'},inplace=True)\nlandmark_counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_img_name = glob.glob('../input/landmark-recognition-2020/train/*/*/*/*')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Visualize some img\nfig = plt.figure(figsize=(16,16))\n\nfor i in range(6):\n    fig.add_subplot(2,3,i+1)\n    img = cv2.imread(train_img_name[i+10])\n    plt.imshow(img)\n    \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[\"filename\"] = train_df.id.str[0]+\"/\"+train_df.id.str[1]+\"/\"+train_df.id.str[2]+\"/\"+train_df.id+\".jpg\"\ntrain_df[\"label\"] = train_df.landmark_id.astype(str)\nprint(train_df.head(3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_class = len(np.unique(train_df.landmark_id.values))\nprint(n_class)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Threshold for number of data each category\n# threshold = 100\n\n# landmark_count_over_threshold = landmark_counts[landmark_counts['count'] > threshold]\n# length_landmark_count_over_threshold = len(landmark_count_over_threshold)\n# print(length_landmark_count_over_threshold)\n\n# landmark_id_values = train_df.landmark_id.values\n# count = Counter(landmark_id_values).most_common(length_landmark_count_over_threshold)\n\n# print(len(count))\n# print(count[0])\n# print(count[-1])\n# keep_landmark_id = []\n\n# for i in count:\n#     keep_landmark_id.append(i[0])\n    \n# train_df = train_df[train_df.landmark_id.isin(keep_landmark_id)]\n# print(len(train_df))\n# print(train_df.head(10))\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_ratio = 0.25\nbatch_size = 128\n\n\ngen = ImageDataGenerator(validation_split=val_ratio)\n\ntrain_gen = gen.flow_from_dataframe(\n    train_df,\n    directory=\"/kaggle/input/landmark-recognition-2020/train/\",\n    x_col=\"filename\",\n    y_col=\"label\",\n    weight_col=None,\n    target_size=(256, 256),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=\"categorical\",\n    batch_size=batch_size,\n    shuffle=True,\n    seed = 44,\n    subset=\"training\",\n    interpolation=\"nearest\",\n    validate_filenames=False)\n    \nval_gen = gen.flow_from_dataframe(\n    train_df,\n    directory=\"/kaggle/input/landmark-recognition-2020/train/\",\n    x_col=\"filename\",\n    y_col=\"label\",\n    weight_col=None,\n    target_size=(256, 256),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=\"categorical\",\n    batch_size=batch_size,\n    shuffle=True,\n    seed = 44,\n    subset=\"validation\",\n    interpolation=\"nearest\",\n    validate_filenames=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def my_model(input_shape,nclass,dropout, learning_rate = 0.001):\n    base_model = MobileNetV2(weights = None, include_top = False)\n    \n    model_input = L.Input(input_shape)\n    x = base_model(model_input)\n    x = L.GlobalAveragePooling2D()(x)\n    \n    y = L.Dense(512,activation='relu')(x)\n    y = L.Dropout(dropout)(y)\n    y = L.Dense(512,activation='relu')(y)\n    y = L.Dropout(dropout)(y)\n    \n    y_h = L.Dense(nclass, activation = 'softmax', name = 'Id')(y)\n    \n    model = Model(inputs=model_input, outputs= y_h)\n    \n    optimizer = Adam(learning_rate=learning_rate)\n    \n    model.compile(loss='categorical_crossentropy', optimizer = optimizer, metrics='accuracy')\n    \n    return model\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = my_model(input_shape = (224,224,3), nclass = n_class, dropout = 0.4)\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# model = load_model(\"../input/my-model/last_model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 1\ntrain_steps = int(len(train_df)*(1-val_ratio))//batch_size\nval_steps = int(len(train_df)*val_ratio)//batch_size\n\nearly_stopping = EarlyStopping(monitor='val_loss', mode='min',patience=6)\nmodel_checkpoint = ModelCheckpoint(\"best_model.h5\", monitor='loss', save_best_only=True, verbose=1)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit_generator(train_gen, steps_per_epoch=train_steps, epochs=epochs,validation_data=val_gen, validation_steps=val_steps, callbacks=[model_checkpoint, early_stopping])\n# history = model.fit_generator(train_gen, steps_per_epoch=train_steps, epochs=epochs,validation_data=val_gen, validation_steps=val_steps)\nmodel.save(\"last_model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save(\"last_model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_model = load_model(\"last_model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/landmark-recognition-2020/sample_submission.csv\") #?? May be here\ntest_df[\"filename\"] = test_df.id.str[0]+\"/\"+test_df.id.str[1]+\"/\"+test_df.id.str[2]+\"/\"+test_df.id+\".jpg\"\nprint(test_df.head(3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_gen = ImageDataGenerator().flow_from_dataframe(\n    test_df,\n    directory=\"/kaggle/input/landmark-recognition-2020/test/\",\n    x_col=\"filename\",\n    y_col=None,\n    weight_col=None,\n    target_size=(256, 256),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=None,\n    batch_size=1,\n    shuffle=True,\n    subset=None,\n    interpolation=\"nearest\",\n    validate_filenames=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_steps = len(test_df)\n\ny_pred_oh = my_model.predict_generator(test_gen, verbose=1, steps = test_steps)\nprint(y_pred_oh.shape)\nprint(y_pred_oh[:2])\n\ny_pred = np.argmax(y_pred_oh, axis=-1)\nprint(y_pred.shape)\nprint(y_pred[:2])\n\ny_prob = np.max(y_pred_oh, axis=-1)\nprint(y_prob.shape)\nprint(y_prob[:2])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_landmark_id = np.unique(train_df.landmark_id.values)\ny_pred_id = [y_landmark_id[Y] for Y in y_pred]\n# print(y_pred)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(test_steps):\n    test_df.loc[i, \"landmarks\"] = str(y_pred_id[i])+\" \"+str(y_prob[i])\ntest_df = test_df.drop(columns=\"filename\")\ntest_df.to_csv(\"/kaggle/working/submission.csv\", index=False)\nprint(test_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}