{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71885,"databundleVersionId":8143495,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-27T14:01:31.708271Z","iopub.execute_input":"2024-05-27T14:01:31.708686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load categories\ntest_categories = pd.read_csv('/kaggle/input/image-matching-challenge-2024/test/categories.csv')\ntrain_categories = pd.read_csv('/kaggle/input/image-matching-challenge-2024/train/categories.csv')\n\n# Load train labels\ntrain_labels = pd.read_csv('/kaggle/input/image-matching-challenge-2024/train/train_labels.csv')\n\n# Inspect the data\nprint(test_categories.head())\nprint(train_categories.head())\nprint(train_labels.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:05:54.703205Z","iopub.execute_input":"2024-05-27T14:05:54.703651Z","iopub.status.idle":"2024-05-27T14:05:54.775754Z","shell.execute_reply.started":"2024-05-27T14:05:54.703618Z","shell.execute_reply":"2024-05-27T14:05:54.774551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import load_img, img_to_array\nimport numpy as np\n\ndef load_image(image_path, target_size=(224, 224)):\n    image = load_img(image_path, target_size=target_size)\n    image = img_to_array(image)\n    image = np.expand_dims(image, axis=0)\n    return image\n\n# Example of loading a test image\nimage_path = '/kaggle/input/image-matching-challenge-2024/test/church/images/00096.png'\nimage = load_image(image_path)\nprint(image.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:06:02.608305Z","iopub.execute_input":"2024-05-27T14:06:02.608761Z","iopub.status.idle":"2024-05-27T14:06:18.687815Z","shell.execute_reply.started":"2024-05-27T14:06:02.608727Z","shell.execute_reply":"2024-05-27T14:06:18.686559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Example of applying data augmentation\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:07:04.429544Z","iopub.execute_input":"2024-05-27T14:07:04.430095Z","iopub.status.idle":"2024-05-27T14:07:04.439610Z","shell.execute_reply.started":"2024-05-27T14:07:04.430031Z","shell.execute_reply":"2024-05-27T14:07:04.438265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\n\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(224, 224, 3)),\n    MaxPooling2D(pool_size=(2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dense(10, activation='softmax')  # Assuming 10 classes\n])\n\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Summary of the model\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:07:15.531253Z","iopub.execute_input":"2024-05-27T14:07:15.531719Z","iopub.status.idle":"2024-05-27T14:07:16.126190Z","shell.execute_reply.started":"2024-05-27T14:07:15.531681Z","shell.execute_reply":"2024-05-27T14:07:16.124833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelBinarizer\nfrom PIL import Image\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:13:57.949531Z","iopub.execute_input":"2024-05-27T14:13:57.950541Z","iopub.status.idle":"2024-05-27T14:13:58.499550Z","shell.execute_reply.started":"2024-05-27T14:13:57.950499Z","shell.execute_reply":"2024-05-27T14:13:58.498378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_images_from_folder(folder, label):\n    images = []\n    labels = []\n    for filename in os.listdir(folder):\n        img_path = os.path.join(folder, filename)\n        try:\n            img = Image.open(img_path).convert('RGB')\n            img = img.resize((128, 128))  # Resize images to a standard size\n            img = np.array(img)\n            images.append(img)\n            labels.append(label)\n        except Exception as e:\n            print(f\"Failed to process {img_path}: {e}\")\n    return images, labels\n\n# List of all categories in the training dataset\ncategories = [\n    'church', \n    'transp_obj_glass_cylinder', \n    'lizard'\n    # Add other categories as needed\n]\n\nall_images = []\nall_labels = []\n\nfor category in categories:\n    folder = f'/kaggle/input/image-matching-challenge-2024/train/{category}/images'\n    images, labels = load_images_from_folder(folder, category)\n    all_images.extend(images)\n    all_labels.extend(labels)\n\n# Convert to numpy arrays\nX = np.array(all_images)\ny = np.array(all_labels)\n\n# Normalize the image data\nX = X / 255.0\n\n# Convert labels to one-hot encoding\nlb = LabelBinarizer()\ny = lb.fit_transform(y)\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:14:05.972202Z","iopub.execute_input":"2024-05-27T14:14:05.972691Z","iopub.status.idle":"2024-05-27T14:15:31.907601Z","shell.execute_reply.started":"2024-05-27T14:14:05.972659Z","shell.execute_reply":"2024-05-27T14:15:31.906465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\ndatagen.fit(X_train)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:15:42.100512Z","iopub.execute_input":"2024-05-27T14:15:42.100903Z","iopub.status.idle":"2024-05-27T14:15:42.257794Z","shell.execute_reply.started":"2024-05-27T14:15:42.100874Z","shell.execute_reply":"2024-05-27T14:15:42.256260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(128, 128, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(len(categories), activation='softmax')\n])\n\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:15:44.099310Z","iopub.execute_input":"2024-05-27T14:15:44.099812Z","iopub.status.idle":"2024-05-27T14:15:44.202507Z","shell.execute_reply.started":"2024-05-27T14:15:44.099767Z","shell.execute_reply":"2024-05-27T14:15:44.201247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(datagen.flow(X_train, y_train, batch_size=32), epochs=10, validation_data=(X_test, y_test))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:15:48.590775Z","iopub.execute_input":"2024-05-27T14:15:48.591212Z","iopub.status.idle":"2024-05-27T14:18:27.542946Z","shell.execute_reply.started":"2024-05-27T14:15:48.591177Z","shell.execute_reply":"2024-05-27T14:18:27.541510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the test set\ntest_loss, test_acc = model.evaluate(X_test, y_test)\nprint(f'Test accuracy: {test_acc:.4f}')\nprint(f'Test loss: {test_loss:.4f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T14:18:57.457324Z","iopub.execute_input":"2024-05-27T14:18:57.458377Z","iopub.status.idle":"2024-05-27T14:18:58.457549Z","shell.execute_reply.started":"2024-05-27T14:18:57.458335Z","shell.execute_reply":"2024-05-27T14:18:58.456169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}