{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This id looks like it includes many dissimilar landmarks as the code below will show.\n\nBased on the evauluation function, it looks like we are rewarded for 1) correctly classifying with high confidence, and 2) penalized when we misclassify.\n\nHow should we handle landmark id(s) like this?\n\nMy gut tells me it may be better to drop this id from the training set since we probably cant classify to this id well and may misclassify to this id as well.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport cv2 as cv\nimport matplotlib.image as mpimg\nfrom matplotlib import pyplot as plt\n\npd.options.display.float_format = '{:.2f}'.format","metadata":{"execution":{"iopub.status.busy":"2021-08-17T14:21:51.772766Z","iopub.execute_input":"2021-08-17T14:21:51.773328Z","iopub.status.idle":"2021-08-17T14:21:52.060512Z","shell.execute_reply.started":"2021-08-17T14:21:51.773187Z","shell.execute_reply":"2021-08-17T14:21:52.059567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels = pd.read_csv('../input/landmark-recognition-2021/train.csv')\n\ntraining_labels['path1'] = training_labels['id'].str.slice(start = 0, stop = 1)\ntraining_labels['path2'] = training_labels['id'].str.slice(start = 1, stop = 2)\ntraining_labels['path3'] = training_labels['id'].str.slice(start = 2, stop = 3)\ntraining_labels['path'] = '../input/landmark-recognition-2021/train/' + training_labels['path1'] + '/' + training_labels['path2'] + '/' + training_labels['path3'] + '/' + training_labels['id'] + '.jpg'\ntraining_labels = training_labels.drop(['path1', 'path2', 'path3'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2021-08-17T14:21:52.062018Z","iopub.execute_input":"2021-08-17T14:21:52.062330Z","iopub.status.idle":"2021-08-17T14:21:56.875596Z","shell.execute_reply.started":"2021-08-17T14:21:52.062282Z","shell.execute_reply":"2021-08-17T14:21:56.874613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"piv = training_labels.pivot_table(index = 'landmark_id', aggfunc=lambda x: len(x.unique()))['id']\npiv.sort_values()","metadata":{"execution":{"iopub.status.busy":"2021-08-17T14:21:56.879351Z","iopub.execute_input":"2021-08-17T14:21:56.879639Z","iopub.status.idle":"2021-08-17T14:22:05.348863Z","shell.execute_reply.started":"2021-08-17T14:21:56.879612Z","shell.execute_reply":"2021-08-17T14:22:05.347878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_landmark = 138982\ntop_landmark_df = training_labels.loc[training_labels['landmark_id'] == top_landmark]\ntop_landmark_sample = top_landmark_df.sample(n = 16, random_state = 2021)\ntop_landmark_path = top_landmark_sample['path']\n\nfig = plt.figure(figsize=(100, 100))\n\nfor i in range(0, 16):\n    fig.add_subplot(4, 4, i+1)\n    plt.imshow(mpimg.imread(top_landmark_path.iloc[i]))\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-08-17T14:22:05.350111Z","iopub.execute_input":"2021-08-17T14:22:05.350554Z","iopub.status.idle":"2021-08-17T14:22:19.089001Z","shell.execute_reply.started":"2021-08-17T14:22:05.350517Z","shell.execute_reply":"2021-08-17T14:22:19.079485Z"},"trusted":true},"execution_count":null,"outputs":[]}]}