{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dataset for all the images ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport tensorflow as tf\nimport keras\nimport keras.layers as L\nimport math\nimport cv2\nfrom keras.utils import Sequence\nfrom keras.preprocessing import image\nfrom random import shuffle\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pip install imutils\n#from imutils import paths\n#image_paths = list(paths.list_images('../input/landmark-recognition-2021/train'))","metadata":{"execution":{"iopub.status.busy":"2021-08-16T08:37:37.546859Z","iopub.execute_input":"2021-08-16T08:37:37.547209Z","iopub.status.idle":"2021-08-16T08:37:48.482781Z","shell.execute_reply.started":"2021-08-16T08:37:37.54718Z","shell.execute_reply":"2021-08-16T08:37:48.481572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_labels=pd.read_csv(\"../input/landmark-recognition-2021/train.csv\")\nGLR_labels.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:39:16.024208Z","iopub.execute_input":"2021-08-16T11:39:16.024767Z","iopub.status.idle":"2021-08-16T11:39:17.428342Z","shell.execute_reply.started":"2021-08-16T11:39:16.024702Z","shell.execute_reply":"2021-08-16T11:39:17.427385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\nle_id = preprocessing.LabelEncoder()\nGLR_labels[\"id_le\"]= le_id.fit_transform(GLR_labels[\"landmark_id\"])\nGLR_labels","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:39:34.510635Z","iopub.execute_input":"2021-08-16T11:39:34.510972Z","iopub.status.idle":"2021-08-16T11:39:34.585546Z","shell.execute_reply.started":"2021-08-16T11:39:34.510943Z","shell.execute_reply":"2021-08-16T11:39:34.584591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:39:37.417803Z","iopub.execute_input":"2021-08-16T11:39:37.418317Z","iopub.status.idle":"2021-08-16T11:39:37.43171Z","shell.execute_reply.started":"2021-08-16T11:39:37.418284Z","shell.execute_reply":"2021-08-16T11:39:37.43075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_paths=[]\npath=\"../input/landmark-recognition-2021/train\"\nfor number in range(len(GLR_labels[\"id\"])):\n    i=GLR_labels[\"id\"][number][0]\n    j=GLR_labels[\"id\"][number][1]\n    k=GLR_labels[\"id\"][number][2]\n    id=GLR_labels[\"id\"][number]\n\n    image_paths.append(path+\"/\"+i+\"/\"+j+\"/\"+k+\"/\"+id+\".jpg\")","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:39:56.959207Z","iopub.execute_input":"2021-08-16T11:39:56.959587Z","iopub.status.idle":"2021-08-16T11:40:48.560914Z","shell.execute_reply.started":"2021-08-16T11:39:56.959555Z","shell.execute_reply":"2021-08-16T11:40:48.559918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_paths[0:10]","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:40:53.145564Z","iopub.execute_input":"2021-08-16T11:40:53.145929Z","iopub.status.idle":"2021-08-16T11:40:53.152802Z","shell.execute_reply.started":"2021-08-16T11:40:53.145892Z","shell.execute_reply":"2021-08-16T11:40:53.151301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_labels[\"Image_paths\"]=image_paths\nGLR_labels","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:40:58.089181Z","iopub.execute_input":"2021-08-16T11:40:58.08956Z","iopub.status.idle":"2021-08-16T11:40:58.106898Z","shell.execute_reply.started":"2021-08-16T11:40:58.089526Z","shell.execute_reply":"2021-08-16T11:40:58.105799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#temp=GLR_labels[\"landmark_id\"].value_counts()\n#temp\n#temp1=temp[temp<50]\n#temp1\n#sum(temp1)\n#for i in range(len(temp1)):\n    #Image_dataset.drop(Image_dataset[Image_dataset['id_le'] == temp1.index[i]].index, inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:24:42.80646Z","iopub.execute_input":"2021-08-16T11:24:42.806849Z","iopub.status.idle":"2021-08-16T11:24:42.840855Z","shell.execute_reply.started":"2021-08-16T11:24:42.806815Z","shell.execute_reply":"2021-08-16T11:24:42.839871Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image_dataset=GLR_labels\nImage_dataset","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:42:00.335122Z","iopub.execute_input":"2021-08-16T11:42:00.33548Z","iopub.status.idle":"2021-08-16T11:42:00.339821Z","shell.execute_reply.started":"2021-08-16T11:42:00.335449Z","shell.execute_reply":"2021-08-16T11:42:00.338901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = []\ntrain_labels = []\n#class_object = annotations['labels']\n\n# loop over the input images\nfor (i, image_path) in enumerate(image_paths):\n    #read image\n    image = cv2.imread(image_path)\n    #make images gray\n    image = cv2.cvtColor(image,cv2.COLOR_BGR2GRAY)\n    #label image using the annotations\n    label = Image_dataset[\"id_le\"][i]\n    # resize image\n    image = cv2.resize(image, (64, 64))\n    # flatten the image\n    #pixels = image.flatten()\n    #Append flattened image to\n    train_images.append(image)\n    train_labels.append(label)\n    if i%100000==0:\n        print(i)\n    #print('Loaded...', '\\U0001F483', 'Image', str(i+1), 'is a', tmp_label)","metadata":{"execution":{"iopub.status.busy":"2021-08-16T10:32:40.737843Z","iopub.execute_input":"2021-08-16T10:32:40.73822Z","iopub.status.idle":"2021-08-16T10:33:36.181766Z","shell.execute_reply.started":"2021-08-16T10:32:40.738182Z","shell.execute_reply":"2021-08-16T10:33:36.180398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_labels.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_name","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_labels.index[GLR_labels['id'] == image_name].tolist()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_labels.head(20)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_labels['landmark_id'].max()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_submission = pd.read_csv('../input/landmark-recognition-2021/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:22:50.171829Z","iopub.execute_input":"2021-08-16T11:22:50.172476Z","iopub.status.idle":"2021-08-16T11:22:50.199791Z","shell.execute_reply.started":"2021-08-16T11:22:50.172432Z","shell.execute_reply":"2021-08-16T11:22:50.198761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_submission.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_submission.head(100)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_submission.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts = GLR_labels.landmark_id.value_counts()\ncounts\n#counts = counts[counts >=50].index #indexing only classes which have atleast 50 samples\n#counts[0:10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts[138982]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\n\nlabel_cat = to_categorical(GLR_labels[\"landmark_id\"])\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_cat.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#GLR_labels = GLR_labels.loc[GLR_labels.landmark_id.isin(counts)]\n#num_classes = counts.shape[0]\nnum_classes = label_cat.shape[1]\n\nprint(num_classes)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLR_labels.shape[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def id2path(idx,is_train=True):\n    path = '../input/landmark-recognition-2021'\n    if is_train:\n        path += '/train/'+idx[0]+'/'+idx[1]+'/'+idx[2]+'/'+idx+'.jpg'\n    else:\n        path += '/test/'+idx[0]+'/'+idx[1]+'/'+idx[2]+'/'+idx+'.jpg'\n    return path\nGLR_labels['file_path'] = GLR_labels['id'].apply(id2path)\nGLR_submission['file_path'] = GLR_submission['id'].apply(id2path,False)","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:22:53.911299Z","iopub.execute_input":"2021-08-16T11:22:53.911717Z","iopub.status.idle":"2021-08-16T11:22:55.918477Z","shell.execute_reply.started":"2021-08-16T11:22:53.911683Z","shell.execute_reply":"2021-08-16T11:22:55.917589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(idx):\n    image = cv2.imread(idx)\n    image = image/255.\n    image = cv2.resize(image,(256,256))\n    return image","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:22:57.878396Z","iopub.execute_input":"2021-08-16T11:22:57.878793Z","iopub.status.idle":"2021-08-16T11:22:57.884152Z","shell.execute_reply.started":"2021-08-16T11:22:57.87876Z","shell.execute_reply":"2021-08-16T11:22:57.883087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_images(landmark_id): #plot images by image_id\n    landmark = GLR_labels[GLR_labels['landmark_id']==landmark_id].head(25)\n    imgs = [read_image(x) for x in landmark['file_path']]\n    _, axs = plt.subplots(5,5, figsize=(12, 12))\n    axs = axs.flatten()\n    for i, (img, ax) in enumerate(zip(imgs, axs)):\n        ax.title.set_text(str(landmark['id'].iloc[i]))\n        ax.imshow(img)\n        ax.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:22:59.760335Z","iopub.execute_input":"2021-08-16T11:22:59.760836Z","iopub.status.idle":"2021-08-16T11:22:59.767545Z","shell.execute_reply.started":"2021-08-16T11:22:59.760804Z","shell.execute_reply":"2021-08-16T11:22:59.766853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_images(138982)","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:25:02.031072Z","iopub.execute_input":"2021-08-16T11:25:02.031521Z","iopub.status.idle":"2021-08-16T11:25:04.064568Z","shell.execute_reply.started":"2021-08-16T11:25:02.03148Z","shell.execute_reply":"2021-08-16T11:25:04.063485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Dataset(Sequence):\n    def __init__(self,idx,y=None,batch_size=128,shuffle=True):\n        self.idx = idx\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        if y is not None:\n            self.is_train=True\n        else:\n            self.is_train=False\n        self.y = y\n    def __len__(self):\n        return math.ceil(len(self.idx)/self.batch_size)\n    def __getitem__(self,ids):\n        batch_ids = self.idx[ids * self.batch_size:(ids + 1) * self.batch_size]\n        if self.y is not None:\n            batch_y = self.y[ids * self.batch_size: (ids + 1) * self.batch_size]\n            \n        list_x = np.array([read_image(x) for x in batch_ids])\n        batch_X = np.stack(list_x)\n        if self.is_train:\n            return batch_X, batch_y\n        else:\n            return batch_X\n    \n    def on_epoch_end(self):\n        if self.shuffle and self.is_train:\n            ids_y = list(zip(self.idx, self.y))\n            shuffle(ids_y)\n            self.idx, self.y = list(zip(*ids_y))","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:25:31.600312Z","iopub.execute_input":"2021-08-16T11:25:31.600759Z","iopub.status.idle":"2021-08-16T11:25:31.612984Z","shell.execute_reply.started":"2021-08-16T11:25:31.60072Z","shell.execute_reply":"2021-08-16T11:25:31.611651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_idx =  GLR_labels['file_path'].values\ny = GLR_labels['landmark_id'].values\ntest_idx = GLR_submission['file_path'].values","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:25:34.574692Z","iopub.execute_input":"2021-08-16T11:25:34.575056Z","iopub.status.idle":"2021-08-16T11:25:34.579901Z","shell.execute_reply.started":"2021-08-16T11:25:34.575023Z","shell.execute_reply":"2021-08-16T11:25:34.578893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_idx","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:25:37.674606Z","iopub.execute_input":"2021-08-16T11:25:37.674984Z","iopub.status.idle":"2021-08-16T11:25:37.68193Z","shell.execute_reply.started":"2021-08-16T11:25:37.674947Z","shell.execute_reply":"2021-08-16T11:25:37.68088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_valid,y_train,y_valid = train_test_split(train_idx,y,test_size=0.05,random_state=42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = Dataset(x_train,y_train)\nvalid_dataset = Dataset(x_valid,y_valid)\ntest_dataset = Dataset(test_idx)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train_dataset))","metadata":{"execution":{"iopub.status.busy":"2021-08-16T11:15:44.513303Z","iopub.execute_input":"2021-08-16T11:15:44.513708Z","iopub.status.idle":"2021-08-16T11:15:44.545779Z","shell.execute_reply.started":"2021-08-16T11:15:44.513676Z","shell.execute_reply":"2021-08-16T11:15:44.544624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp=train_dataset[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset[0][0].shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset[0][1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB1\nmodel = EfficientNetB1(weights='imagenet')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U efficientnet\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import efficientnet.keras as efn\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.Sequential(\n        [efn.EfficientNetB0(include_top=False,input_shape=(256,256,3),weights='imagenet'),\n        L.GlobalAveragePooling2D(),\n        L.Dense(128,activation='relu'),\n        L.Dense(64,activation='relu'),\n        L.Dense(num_classes, activation='sigmoid')])\nmodel.summary()\nmodel.compile(optimizer=keras.optimizers.Adam(learning_rate=0.001),\n              loss=keras.losses.SparseCategoricalCrossentropy(), metrics=[keras.metrics.SparseCategoricalAccuracy()])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_dataset,epochs=1,validation_data=valid_dataset)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}