{"cells":[{"metadata":{},"cell_type":"markdown","source":"#### This notebook was written for beginners.\n#### I want to perform data analysis. Check the number of train data and test data, and find out the distribution of each class. And print the image of each class.","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras import Model\nfrom tensorflow.keras.layers import Flatten, Dense, GlobalAveragePooling2D\nfrom tensorflow.keras import optimizers\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras import callbacks\nfrom tqdm import tqdm\nimport cv2\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nimport seaborn as sns\n\n!pip install efficientnet\nimport efficientnet.tfkeras as efn","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Dataset parameters:\nINPUT_DIR = os.path.join('..', 'input')\n\nDATASET_DIR = os.path.join(INPUT_DIR, 'landmark-recognition-2020')\nTRAIN_IMAGE_DIR = os.path.join(DATASET_DIR, 'train')\nTEST_IMAGE_DIR = os.path.join(DATASET_DIR, 'test')\nTRAIN_CSV_PATH = os.path.join(DATASET_DIR, 'train.csv')\n\nSUBMISSION_PATH = os.path.join(DATASET_DIR, 'sample_submission.csv')\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Train image size","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN_DF = pd.read_csv(TRAIN_CSV_PATH)\nprint(f'TRAIN SIZE : {len(TRAIN_DF)}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN_DF['landmark_id'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN_DF.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Count per landmark","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df2 = pd.DataFrame(TRAIN_DF['landmark_id'].value_counts())\ntrain_df2.reset_index(inplace=True)\ntrain_df2.columns = ['landmark_id','count']\ntrain_df2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib as mpl\nimport matplotlib.pylab as plt\n\nplt.figure(figsize = (14, 10))\nplt.title('Top 20 landmarks')\ng = sns.barplot(x=\"landmark_id\", y=\"count\", data=train_df2[:20], palette=\"pastel\")\n\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Test Image Size","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"TEST_DF = pd.read_csv(SUBMISSION_PATH)\nprint(f'TEST SIZE : {len(TEST_DF)}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = []\ndata = []\nfor i in range(TRAIN_DF.shape[0]):\n    data.append(TRAIN_IMAGE_DIR + '/' + TRAIN_DF['id'].iloc[i][0] + '/' + TRAIN_DF['id'].iloc[i][1] + '/' + TRAIN_DF['id'].iloc[i][2] + '/' + TRAIN_DF['id'].iloc[i]+'.jpg')\n    labels.append(TRAIN_DF['landmark_id'].iloc[i])\n\ndf = pd.DataFrame(data)\ndf.columns = ['images']\ndf['target'] = labels \n    \n\ntest_data = []\nfor i in range(TEST_DF.shape[0]):\n    test_data.append(TEST_IMAGE_DIR + '/' + TEST_DF['id'].iloc[i][0] + '/' + TEST_DF['id'].iloc[i][1] + '/' + TEST_DF['id'].iloc[i][2] + '/' + TEST_DF['id'].iloc[i]+'.jpg')\n\ndf_test = pd.DataFrame(test_data)\ndf_test.columns = ['images']    \n\nX_train, X_val, y_train, y_val = train_test_split(df['images'], df['target'], test_size=0.1, random_state=1234)\n\ntrain=pd.DataFrame(X_train)\ntrain.columns=['images']\ntrain['target']=y_train\n\nvalidation=pd.DataFrame(X_val)\nvalidation.columns=['images']\nvalidation['target']=y_val","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Select landmark id & show data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df2 = df[df['target']==66]\ndf2","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Visualize image of selected randmark id\n#### Double-click the image to see a larger image.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfig = plt.figure(figsize=(70,70))\nrows = 1\ncols = len(df2)\n\n\nfor i in range(len(df2)):\n\n    img1 = cv2.imread(df2['images'].iloc[i])\n\n    ax1 = fig.add_subplot(rows, len(df2), i+1)\n    ax1.imshow(cv2.cvtColor(img1, cv2.COLOR_BGR2RGB))\n    ax1.set_title(df2['target'].iloc[i])\n    ax1.axis(\"off\")\n\n \nplt.show()\nprint(f'size : {len(df2)}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model - efficientNet B3","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_model(num_classes=None, input_size=224):\n    model = efn.EfficientNetB3(weights='noisy-student', include_top=False, input_shape=(input_size, input_size, 3))\n    x = GlobalAveragePooling2D()(model.output)\n    output = Dense(num_classes, activation='softmax')(x)\n    model = Model(model.input, output)\n\n    return model\n\n\ndef lr_scheduler(epoch, lr):\n    if epoch == 3 or epoch == 8:\n        return lr * 0.1\n    else:\n        return lr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1./255,\n                                   zoom_range=[1.0, 1.2],\n                                   brightness_range=[0.8, 1.1],\n                                   width_shift_range=0.1,\n                                   height_shift_range=0.1,\n                                  )\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    train,\n    x_col='images',\n    y_col='target',\n    target_size=(224, 224),\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    class_mode='raw')\n\nvalidation_generator = val_datagen.flow_from_dataframe(\n    validation,\n    x_col='images',\n    y_col='target',\n    target_size=(224, 224),\n    shuffle=False,\n    batch_size=BATCH_SIZE,\n    class_mode='raw')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = create_model(8226, 224)\n#model.summary()\n\nopt = optimizers.Adam(lr=1e-4)\nmodel.compile(\n    loss = 'sparse_categorical_crossentropy',\n    metrics=['sparse_categorical_accuracy'],\n    optimizer=opt)\n\nnb_train_steps = train.shape[0]//BATCH_SIZE\nnb_val_steps = validation.shape[0]//BATCH_SIZE\n\nlearning_rate_callback = tf.keras.callbacks.LearningRateScheduler(lr_scheduler)\n\nhistory = model.fit_generator(\n    train_generator,\n    steps_per_epoch=nb_train_steps,\n    epochs=EPOCHS,\n    validation_data=validation_generator,\n    callbacks=[learning_rate_callback],\n    validation_steps=nb_val_steps)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Predict","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"for path in tqdm(df_test['images']):\n    img= cv2.imread(str(path))\n    img = cv2.resize(img, (224,224))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = img.astype(np.float32)/255.\n    img=np.reshape(img,(1,224,224,3))\n    prediction=model.predict(img)\n    target.append(prediction[0][0])\n\nsubmission['target']=target\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}