{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n #   for filename in filenames:\n  #      print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\nimport keras\nimport cv2\nfrom matplotlib import pyplot as plt\nimport os\nimport random\nfrom PIL import Image\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nsample_path = r\"/kaggle/input/landmark-recognition-2020/sample_submission.csv\"\ntrain_path = r\"/kaggle/input/landmark-recognition-2020/train.csv\"\nbase_path = r\"/kaggle/input/landmark-recognition-2020/train\"\ntest_path = r\"/kaggle/input/landmark-recognition-2020/test\"\n\n\ndf = pd.read_csv(\"../input/landmark-recognition-2020/train.csv\")# Read the CSV file containing the training labels etc.\n\ndf = df.loc[:15000,:]\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Size of training data:\", df.shape)\n#Count how many unique landmarks there are, that is to say the amount of classes\nprint(\"Number of unique classes:\", num_classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.DataFrame(df['landmark_id'].value_counts()) #make data frame that is easier to use\n#index the data frame\ndata.reset_index(inplace=True) \ndata.columns=['landmark_id','count']\n\n\ndata = data.iloc[80:200]\nnum_classes = len(data[\"landmark_id\"].unique())\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plt.hist(data['count'],100,range = (0,944),label = 'test')#Histogram of the distribution\n# plt.xlabel(\"Amount of images\")\n# plt.ylabel(\"Occurences\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""},{"metadata":{"trusted":true},"cell_type":"code","source":"# print(\"Amount of classes with five and less datapoints:\", (data['count'].between(0,5)).sum()) \n\n# print(\"Amount of classes with with between five and 10 datapoints:\", (data['count'].between(5,10)).sum())\n\n# n = plt.hist(df[\"landmark_id\"],bins=df[\"landmark_id\"].unique())\n# freq_info = n[0]\n\n# plt.xlim(0,data['landmark_id'].max())\n# plt.ylim(0,data['count'].max())\n# plt.xlabel('Landmark ID')\n# plt.ylabel('Number of images')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlencoder = LabelEncoder()\nlencoder.fit(data[\"landmark_id\"])\n\ndef encode_label(lbl):\n    return lencoder.transform(lbl)\n    \ndef decode_label(lbl):\n    return lencoder.inverse_transform(lbl)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_image_from_number(num):\n    fname, label = df.loc[num,:]\n    fname = fname + \".jpg\"\n    f1 = fname[0]\n    f2 = fname[1]\n    f3 = fname[2]\n    path = os.path.join(f1,f2,f3,fname)\n    im = cv2.imread(os.path.join(base_path,path))\n    return im, label\n\n### Function used for processing the data, fitted into a data generator.\ndef get_image_from_number(num, df):\n    fname, label = df.iloc[num,:]\n    fname = fname + \".jpg\"\n    f1 = fname[0]\n    f2 = fname[1]\n    f3 = fname[2]\n    path = os.path.join(f1,f2,f3,fname)\n    im = cv2.imread(os.path.join(base_path,path))\n    return im, label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications import VGG19\nfrom keras.layers import *\nfrom keras import Sequential\n\nsource_model = VGG19(weights=None)\n#new_layer = Dense(num_classes, activation=activations.softmax, name='prediction')\ndrop_layer = Dropout(0.5)\ndrop_layer2 = Dropout(0.5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nfor layer in source_model.layers[:-1]: # go through until last layer\n    if layer == source_model.layers[-25]:\n        model.add(BatchNormalization())\n    model.add(layer)\n    #if layer == source_model.layers[-3]:\n     #   model.add(drop_layer)\n    #model.add(drop_layer2)\nmodel.add(Dense(num_classes, activation=\"softmax\"))\nmodel.summary()\n\n\nopt1 = keras.optimizers.RMSprop(learning_rate = 0.0001, momentum = 0.09)\nopt2 = keras.optimizers.Adam(learning_rate=0.001, beta_1=0.9, beta_2=0.999, epsilon=1e-07)\nmodel.compile(optimizer=opt1,\n             loss=\"sparse_categorical_crossentropy\",\n             metrics=[\"accuracy\"])\n\n#sgd = SGD(lr=learning_rate, decay=decay_speed, momentum=momentum, nesterov=True)\n# rms = keras.optimizers.RMSprop(lr=learning_rate, momentum=momentum)\n# model.compile(optimizer=rms,\n#               loss=loss_function,\n#               metrics=[\"accuracy\"])\n# print(\"Model compiled! \\n\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def image_reshape(im, target_size):\n    return cv2.resize(im, target_size)\n    \ndef get_batch(dataframe,start, batch_size):\n    image_array = []\n    label_array = []\n    \n    end_img = start+batch_size\n    if end_img > len(dataframe):\n        end_img = len(dataframe)\n\n    for idx in range(start, end_img):\n        n = idx\n        im, label = get_image_from_number(n, dataframe)\n        im = image_reshape(im, (224, 224)) / 255.0\n        image_array.append(im)\n        label_array.append(label)\n        \n    label_array = encode_label(label_array)\n    return np.array(image_array), np.array(label_array)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.utils.np_utils import to_categorical\nname = []\ncat = []\nlsit  = list(data['landmark_id'])\n\nfor i in range(len(df)):\n    if df.iloc[i,1] not in lsit:\n        #print(\"YES\")\n        continue\n        \n    else:\n        name.append(df.iloc[i,0])\n        cat.append(df.iloc[i,1])\ndataF=pd.DataFrame(list(zip(name, cat)),columns=['FILE','CLASS'])\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del name\ndel cat\ndel data\ndel df\ndel lsit","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(num_classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#dataF['CLASS'] =  to_categorical(dataF['CLASS'], num_classes=num_classes)\nprint(dataF.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import sys\nbatch_size = 200\nepoch_shuffle = True\nweight_classes = True\nepochs = 50\n\n# Split train data up into 80% and 20% validation\n\n\ntrain,validate = np.split(dataF.sample(frac=1), [int(.8*len(dataF))])\nprint(\"Training on:\", len(train), \"samples\")\nprint(\"Validation on:\", len(validate), \"samples\")\n\n    \nfor e in range(epochs):\n    print(\"Epoch: \", str(e+1) + \"/\" + str(epochs))\n    if epoch_shuffle:\n        train = train.sample(frac = 1)\n    for it in range(int(np.ceil(len(train)/batch_size))):\n        #print(\"Current batch number:\",it)\n        X_train, y_train = get_batch(train, it*batch_size, batch_size)\n        training_datagen = ImageDataGenerator(rotation_range=40, \n                        width_shift_range=0.2, \n                        height_shift_range=0.2, \n                        zoom_range=0.2, \n                        horizontal_flip=True, \n                        vertical_flip=True,\n                        shear_range=0.2) \n\n        train_generator = training_datagen.flow(X_train, y_train)\n        training_datagen.fit(X_train)\n        del X_train\n        del y_train\n        history = model.fit_generator(train_generator)\n        #model.train_on_batch(X_train, y_train)\n        \nmodel.save(\"Model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### Test on training set\nbatch_size = 20\n\nerrors = 0\ngood_preds = []\nbad_preds = []\n\n\nfor it in range(int(np.ceil(len(validate)/batch_size))):\n\n    \n    X_train, y_train = get_batch(validate, it*batch_size, batch_size)\n    \n    result = model.predict(X_train)\n    cla = np.argmax(result, axis=1)\n    for idx, res in enumerate(result):\n        print(\"Class:\", cla[idx], \"- Confidence:\", np.round(res[cla[idx]],2), \"- GT:\", y_train[idx])\n        if cla[idx] != y_train[idx]:\n            errors = errors + 1\n            bad_preds.append([batch_size*it + idx, cla[idx], res[cla[idx]]])\n        else:\n            good_preds.append([batch_size*it + idx, cla[idx], res[cla[idx]]])\n\nprint(\"Errors: \", errors, \"Acc:\", np.round(100*(len(validate)-errors)/len(validate),2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import precision_recall_fscore_support as score\nprecision, recall, fscore, support = score(y_train, result)\n\nprint('precision: {}'.format(precision))\nprint('recall: {}'.format(recall))\nprint('fscore: {}'.format(fscore))\nprint('support: {}'.format(support))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"good_preds = np.array(good_preds)\ngood_preds = np.array(sorted(good_preds, key = lambda x: x[2], reverse=True))\n\nprint(\"5 images where classification went well:\")\nfig=plt.figure(figsize=(16, 16))\nfor i in range(2,6):\n    n = int(good_preds[i,0])\n    img, lbl = get_image_from_number(n, validate)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    fig.add_subplot(1,6, i)\n    plt.imshow(img)\n    lbl2 = np.array(int(good_preds[i,1])).reshape(1,1)\n    sample_cn = list(dataF['CLASS']).count(lbl)\n    plt.title(\"Label: \" + str(lbl) + \"\\nClassified as: \" + str(decode_label(lbl2)) + \"\\nSamples in class \" + str(lbl) + \": \" + str(sample_cn))\n    plt.axis('off')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bad_preds = np.array(bad_preds)\nbad_preds = np.array(sorted(bad_preds, key = lambda x: x[2], reverse=True))\n\nprint(\"5 images where classification failed:\")\nfig=plt.figure(figsize=(16, 16))\nfor i in range(1,6):\n    n = int(bad_preds[i,0])\n    img, lbl = get_image_from_number(n, validate)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    fig.add_subplot(1, 5, i)\n    plt.imshow(img)\n    lbl2 = np.array(int(good_preds[i,1])).reshape(1,1)\n    sample_cn = list(dataF['CLASS']).count(lbl)\n    plt.title(\"Label: \" + str(lbl) + \"\\nClassified as: \" + str(decode_label(lbl2)) + \"\\nSamples in class \" + str(lbl) + \": \" + str(sample_cn))\n    plt.axis('off')\n    \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np \nimport matplotlib.pyplot as plt\nimport os\nfrom urllib.request import urlopen,urlretrieve\nfrom PIL import Image\nfrom tqdm import tqdm_notebook\n%matplotlib inline\nfrom sklearn.utils import shuffle\nimport cv2\n#import tensorflow\n#from resnets_utils import *\n\nfrom keras.models import load_model\nfrom sklearn.datasets import load_files   \nfrom keras.utils import np_utils\nfrom glob import glob\nfrom keras import applications\nfrom keras.preprocessing.image import ImageDataGenerator \nfrom keras import optimizers\nfrom keras.models import Sequential,Model,load_model\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D,GlobalAveragePooling2D\nfrom keras.callbacks import TensorBoard,ReduceLROnPlateau,ModelCheckpoint","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train,validate = np.split(dataF.sample(frac=1), [int(.7*len(dataF))])\nprint(\"Training on:\", len(train), \"samples\")\nprint(\"Validation on:\", len(validate), \"samples\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_height,img_width = 224,224\n\n#If imagenet weights are being loaded, \n#input must have a static square shape (one of (128, 128), (160, 160), (192, 192), or (224, 224))\nbase_model = applications.resnet50.ResNet50(weights= None, include_top=False, input_shape= (img_height,img_width,3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dropout(0.7)(x)\npredictions = Dense(num_classes, activation= 'softmax')(x)\nmodel = Model(inputs = base_model.input, outputs = predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.optimizers import SGD, Adam\n# sgd = SGD(lr=lrate, momentum=0.9, decay=decay, nesterov=False)\nadam = Adam(lr=0.0001)\nmodel.compile(optimizer= adam, loss='sparse_categorical_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndel X_test\ndel Y_test\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, Y_train = get_batch(train, 0, len(train))\ndel(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(X_train, Y_train, epochs = 60, batch_size = 48)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_test, Y_test = get_batch(validate,0, len(validate)-200)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = model.evaluate(X_test, Y_test)\n\nprint (\"Loss = \" + str(preds[0]))\nprint (\"Test Accuracy = \" + str(preds[1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"t = model.predict(X_test)\nprint(t)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 16\n\nerrors = 0\ngood_preds = []\nbad_preds = []\ncla = np.argmax(t,axis=1)\nfor idx, res in enumerate(t):\n        print(\"Class:\", cla[idx], \"- Confidence:\", np.round(res[cla[idx]],2), \"- GT:\", Y_test[idx])\n        if cla[idx] != Y_test[idx]:\n            errors = errors + 1\n            bad_preds.append([idx, cla[idx], res[cla[idx]]])\n        else:\n            good_preds.append([idx, cla[idx], res[cla[idx]]])\n\nprint(\"Errors: \", errors, \"Acc:\", np.round(100*(len(validate)-errors)/len(validate),2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.layers import  Flatten, Dense, Dropout\nfrom keras.applications import VGG16\nfrom keras.models import Model\nfrom keras import optimizers\nfrom keras.optimizers import Adam\nfrom keras.layers import Dense, GlobalAveragePooling2D\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization, AveragePooling2D, GlobalAveragePooling2D\nvgg16_model = VGG16(weights = 'imagenet', include_top = False,input_shape=(224,224,3))\nx = vgg16_model.output\nx = GlobalAveragePooling2D()(x)\nx = BatchNormalization()(x)\nx = Dropout(0.5)(x)\nx = Dense(256, activation='relu')(x)\nx = BatchNormalization()(x)\nx = Dropout(0.5)(x)\n\npredictions = Dense(num_classes, activation = 'softmax')(x)\nmodel2 = Model(vgg16_model.input,predictions)\nfor layer in vgg16_model.layers:\n    layer.trainable = False\noptimizer = Adam(lr=0.0002)\nmodel2.compile(loss='sparse_categorical_crossentropy', optimizer=optimizer, metrics=['accuracy'])\n\nmodel2.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import keras_preprocessing\nfrom keras_preprocessing import image\nfrom keras_preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"training_datagen = ImageDataGenerator(rotation_range=40, \n                        width_shift_range=0.2, \n                        height_shift_range=0.2, \n                        zoom_range=0.2, \n                        horizontal_flip=True, \n                        vertical_flip=True,\n                        shear_range=0.2) \n\ntrain_generator = training_datagen.flow(X_train, Y_train,batch_size=64)\ntraining_datagen.fit(X_train)\ndel X_train\ndel Y_train\nfrom keras import callbacks\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filepath=\"Best1.hdf5\"\ncheckpoint = callbacks.ModelCheckpoint(filepath, monitor='val_loss',save_best_only=True, mode='min',verbose=1)\ncallbacks_list = [checkpoint]\n\nhistory = model2.fit_generator(train_generator, steps_per_epoch=35, epochs=100,\n                              validation_data=(X_test, Y_test),validation_steps=50,callbacks=callbacks_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs = range(len(acc))\nplt.plot(epochs, acc, 'r', label='Training accuracy')\nplt.plot(epochs, val_acc, 'b', label='Validation accuracy')\nplt.title('Training and validation accuracy')\nplt.legend(loc=0)\nplt.figure()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model2.load_weights(\"Best1.hdf5\")\nscore = model2.evaluate(X_test, Y_test ,verbose=1)\nprint('Test Loss:', score[0])\nprint('Test accuracy:', score[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#200 epochs\nhistory = model2.fit_generator(train_generator, steps_per_epoch=35, epochs=80,\n                              validation_data=(X_test, Y_test),validation_steps=50,callbacks=callbacks_list)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}