{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-01T16:54:40.042608Z","iopub.execute_input":"2023-05-01T16:54:40.043296Z","iopub.status.idle":"2023-05-01T16:54:41.243202Z","shell.execute_reply.started":"2023-05-01T16:54:40.043242Z","shell.execute_reply":"2023-05-01T16:54:41.242162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Set the current directory to the variable\nroot_dir = os.getcwd()\ninput_dir = '/kaggle/input'\n# Changed the directory into input directory\nos.chdir(input_dir)\nnew_dir = os.getcwd()\nprint('New working directory:', new_dir)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:54:41.245229Z","iopub.execute_input":"2023-05-01T16:54:41.245976Z","iopub.status.idle":"2023-05-01T16:54:41.253164Z","shell.execute_reply.started":"2023-05-01T16:54:41.245937Z","shell.execute_reply":"2023-05-01T16:54:41.251842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_files = []\nfor dirpath, dirnames, filenames in os.walk(new_dir):\n        for filename in filenames:\n            if filename.endswith('.jpeg'):\n                filepath = os.path.join(root_dir, filename)\n                # print(filepath)\n                image_files.append(filepath)\n#                 filepath                \n\nprint(len(image_files))\nprint(image_files)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:12.834985Z","iopub.execute_input":"2023-04-29T19:38:12.835397Z","iopub.status.idle":"2023-04-29T19:38:14.787819Z","shell.execute_reply.started":"2023-04-29T19:38:12.835359Z","shell.execute_reply":"2023-04-29T19:38:14.786914Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to read image files\nimport cv2\n# cv2 reads the image data from a file into a NumPy array\nimages = []\nfor file in image_files:\n    image = cv2.imread(file)\n    images.append(image)\n\nprint(len(images))\n# print('/kaggle/input/image-matching-challenge-2023/train/haiper/chairs/images/image_004.jpeg')","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:14.789541Z","iopub.execute_input":"2023-04-29T19:38:14.790188Z","iopub.status.idle":"2023-04-29T19:38:14.831135Z","shell.execute_reply.started":"2023-04-29T19:38:14.790149Z","shell.execute_reply":"2023-04-29T19:38:14.829859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the dataset into dataframe","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/image-matching-challenge-2023/train/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:54:43.733408Z","iopub.execute_input":"2023-05-01T16:54:43.733928Z","iopub.status.idle":"2023-05-01T16:54:43.758142Z","shell.execute_reply.started":"2023-05-01T16:54:43.733878Z","shell.execute_reply":"2023-05-01T16:54:43.757064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploration Data Analysis","metadata":{}},{"cell_type":"code","source":"#first 5 data\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:54:46.877707Z","iopub.execute_input":"2023-05-01T16:54:46.879458Z","iopub.status.idle":"2023-05-01T16:54:46.914655Z","shell.execute_reply.started":"2023-05-01T16:54:46.879383Z","shell.execute_reply":"2023-05-01T16:54:46.913132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.size","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:18.102567Z","iopub.execute_input":"2023-04-29T19:38:18.103393Z","iopub.status.idle":"2023-04-29T19:38:18.113404Z","shell.execute_reply.started":"2023-04-29T19:38:18.103358Z","shell.execute_reply":"2023-04-29T19:38:18.112169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_path'][0]","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:14.872232Z","iopub.execute_input":"2023-04-29T19:38:14.872981Z","iopub.status.idle":"2023-04-29T19:38:14.881291Z","shell.execute_reply.started":"2023-04-29T19:38:14.872946Z","shell.execute_reply":"2023-04-29T19:38:14.880278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#number of rows and columns\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:14.882667Z","iopub.execute_input":"2023-04-29T19:38:14.883131Z","iopub.status.idle":"2023-04-29T19:38:14.894120Z","shell.execute_reply.started":"2023-04-29T19:38:14.883083Z","shell.execute_reply":"2023-04-29T19:38:14.893009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:14.896896Z","iopub.execute_input":"2023-04-29T19:38:14.897940Z","iopub.status.idle":"2023-04-29T19:38:14.910948Z","shell.execute_reply.started":"2023-04-29T19:38:14.897905Z","shell.execute_reply":"2023-04-29T19:38:14.909834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#summarize statistically\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:14.912492Z","iopub.execute_input":"2023-04-29T19:38:14.913380Z","iopub.status.idle":"2023-04-29T19:38:14.938790Z","shell.execute_reply.started":"2023-04-29T19:38:14.913332Z","shell.execute_reply":"2023-04-29T19:38:14.937700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe(include= 'all')","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:14.942391Z","iopub.execute_input":"2023-04-29T19:38:14.942738Z","iopub.status.idle":"2023-04-29T19:38:14.963901Z","shell.execute_reply.started":"2023-04-29T19:38:14.942705Z","shell.execute_reply":"2023-04-29T19:38:14.962557Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check null values\ndf.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:14.965581Z","iopub.execute_input":"2023-04-29T19:38:14.966060Z","iopub.status.idle":"2023-04-29T19:38:14.974762Z","shell.execute_reply.started":"2023-04-29T19:38:14.966014Z","shell.execute_reply":"2023-04-29T19:38:14.973704Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:14.976275Z","iopub.execute_input":"2023-04-29T19:38:14.976663Z","iopub.status.idle":"2023-04-29T19:38:14.998901Z","shell.execute_reply.started":"2023-04-29T19:38:14.976620Z","shell.execute_reply":"2023-04-29T19:38:14.997658Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check duplicates\ndf.duplicated()","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:15.000256Z","iopub.execute_input":"2023-04-29T19:38:15.000581Z","iopub.status.idle":"2023-04-29T19:38:15.014497Z","shell.execute_reply.started":"2023-04-29T19:38:15.000550Z","shell.execute_reply":"2023-04-29T19:38:15.013461Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#grouping rows by values\ndf.groupby('rotation_matrix').mean()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:15.016270Z","iopub.execute_input":"2023-04-29T19:38:15.016788Z","iopub.status.idle":"2023-04-29T19:38:15.036312Z","shell.execute_reply.started":"2023-04-29T19:38:15.016742Z","shell.execute_reply":"2023-04-29T19:38:15.035074Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = df.corr()\ncorr","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:15.037830Z","iopub.execute_input":"2023-04-29T19:38:15.038494Z","iopub.status.idle":"2023-04-29T19:38:15.047785Z","shell.execute_reply.started":"2023-04-29T19:38:15.038437Z","shell.execute_reply":"2023-04-29T19:38:15.046611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize heatmap for detecting null values","metadata":{}},{"cell_type":"code","source":"plt.subplots(figsize=(10,10))\nsns.heatmap(df.isnull(), yticklabels=False, cbar=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:15.049528Z","iopub.execute_input":"2023-04-29T19:38:15.049939Z","iopub.status.idle":"2023-04-29T19:38:15.253399Z","shell.execute_reply.started":"2023-04-29T19:38:15.049896Z","shell.execute_reply":"2023-04-29T19:38:15.252285Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Heat map is used to check that null values are there. No null values in the dataset**","metadata":{}},{"cell_type":"markdown","source":"# Get Sample Images","metadata":{}},{"cell_type":"code","source":"#imported tensorflow to read images\nimport tensorflow as tf\nimport os.path\npath = \"/kaggle/input/image-matching-challenge-2023/train/\"","metadata":{"execution":{"iopub.status.busy":"2023-05-01T17:10:25.395148Z","iopub.execute_input":"2023-05-01T17:10:25.396671Z","iopub.status.idle":"2023-05-01T17:10:34.754582Z","shell.execute_reply.started":"2023-05-01T17:10:25.396605Z","shell.execute_reply":"2023-05-01T17:10:34.753099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loaded image into variable\n#number of image paths 327\nimg = []\nfor i in range(327):\n    load_image = tf.keras.preprocessing.image.load_img(path + df['image_path'][i])\n    img.append(load_image)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:15.264280Z","iopub.execute_input":"2023-04-29T19:38:15.264612Z","iopub.status.idle":"2023-04-29T19:38:15.917386Z","shell.execute_reply.started":"2023-04-29T19:38:15.264579Z","shell.execute_reply":"2023-04-29T19:38:15.916157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get all images for given structure","metadata":{}},{"cell_type":"code","source":"#create dataframe that has only given images\nscene = df[df['scene'].values ==\"dioscuri\"]\nscene","metadata":{"execution":{"iopub.status.busy":"2023-05-01T17:09:32.834208Z","iopub.execute_input":"2023-05-01T17:09:32.834757Z","iopub.status.idle":"2023-05-01T17:09:32.853001Z","shell.execute_reply.started":"2023-05-01T17:09:32.834711Z","shell.execute_reply":"2023-05-01T17:09:32.852080Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get all image_path into an array\nimport numpy as np\nimg_paths = np.array(scene['image_path'])\nimg_paths","metadata":{"execution":{"iopub.status.busy":"2023-05-01T17:17:37.535983Z","iopub.execute_input":"2023-05-01T17:17:37.536505Z","iopub.status.idle":"2023-05-01T17:17:37.548028Z","shell.execute_reply.started":"2023-05-01T17:17:37.536461Z","shell.execute_reply":"2023-05-01T17:17:37.546242Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load images into an array\nimg = []\nfor img_path in img_paths:\n    load_image = tf.keras.preprocessing.image.load_img(path + img_path)\n    img.append(load_image)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T17:23:25.239540Z","iopub.execute_input":"2023-05-01T17:23:25.239965Z","iopub.status.idle":"2023-05-01T17:23:25.463959Z","shell.execute_reply.started":"2023-05-01T17:23:25.239927Z","shell.execute_reply":"2023-05-01T17:23:25.462433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img[10]","metadata":{"execution":{"iopub.status.busy":"2023-05-01T17:26:15.418780Z","iopub.execute_input":"2023-05-01T17:26:15.419189Z","iopub.status.idle":"2023-05-01T17:26:15.644187Z","shell.execute_reply.started":"2023-05-01T17:26:15.419155Z","shell.execute_reply":"2023-05-01T17:26:15.642462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img[22]","metadata":{"execution":{"iopub.status.busy":"2023-05-01T17:18:31.714572Z","iopub.execute_input":"2023-05-01T17:18:31.715039Z","iopub.status.idle":"2023-05-01T17:18:32.046266Z","shell.execute_reply.started":"2023-05-01T17:18:31.714990Z","shell.execute_reply":"2023-05-01T17:18:32.044642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_files = []\n# for dirpath, dirnames, filenames in os.walk(root_dir):\n#         for filename in filenames:\n#             if filename.endswith('.jpeg'):\n#                 filepath = os.path.join(data_dir, filename)\n#                 image_files.append(filepath)\n#                 filepath                \n# len(image_files)   \n\n# for root, subfolders, filenames in os.walk(root_dir):\n#     for filename in filenames:\n#         filepath = root + \"/\" + filename\n#         print(filepath)\n        # do stuff with filepath","metadata":{"execution":{"iopub.status.busy":"2023-04-29T19:38:18.080811Z","iopub.execute_input":"2023-04-29T19:38:18.081548Z","iopub.status.idle":"2023-04-29T19:38:18.086953Z","shell.execute_reply.started":"2023-04-29T19:38:18.081497Z","shell.execute_reply":"2023-04-29T19:38:18.085793Z"},"trusted":true},"execution_count":null,"outputs":[]}]}