{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":98450,"databundleVersionId":11749951,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sbn\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv3D, MaxPooling3D, Flatten, Dense, Dropout\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.decomposition import PCA\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-11T12:10:34.877109Z","iopub.execute_input":"2025-05-11T12:10:34.877417Z","iopub.status.idle":"2025-05-11T12:10:48.400993Z","shell.execute_reply.started":"2025-05-11T12:10:34.877395Z","shell.execute_reply":"2025-05-11T12:10:48.400444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DataProcessing:\n    def __init__(self):\n        # self.read_path = read_path\n        # self.write_path = write_path\n        pass\n\n    def readData(self, read_path:str) -> pd.DataFrame:\n        df = pd.read_csv(read_path)\n        return df\n\n    def writeData(self, df:pd.DataFrame, write_path:str) -> None:\n        df.to_csv(write_path, index=False)\n\n\n    def dataLoader(self, data_labels:pd.DataFrame, \n                   folder_path:str) -> tuple:\n        train_lis = []\n        test_lis = []\n        for file in os.listdir(folder_path):\n            try:\n                if file in list(test_labels['id']):\n                    test_lis.append(np.load(os.path.join(folder_path, file)))\n                else:\n                    train_lis.append(np.load(os.path.join(folder_path, file)))\n            except:\n                print(file, \" Missed\")\n        return train_lis, test_lis\n\n    def bandsPCA(self, image:np.ndarray) -> list:\n        h, w, c = image.shape\n        pca = PCA(n_components=3)\n        best_comps = pca.fit_transform(image.reshape(h * w, c))\n        reshaped = best_comps.reshape(h, w, 3)\n        max_val, min_val = reshaped.max(), reshaped.min()\n        if max_val == min_val:\n            max_val = max_val + 0.00001\n        result = ((reshaped - max_val) / (max_val - min_val) * 255).astype(np.uint8)\n        return result\\\n\n    def submission(self, test_df:pd.DataFrame, model:keras.models.Model, test_data:list, write_path:str):\n            sub = test_df[['id']]\n            predictions = model.predict(test_data)\n            predictions = [np.argmax(y_pred) for y_pred in predictions]\n            sub['label'] = predictions\n            print(sub.shape, len(predictions))\n            sub.to_csv(write_path, index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-11T12:10:48.402157Z","iopub.execute_input":"2025-05-11T12:10:48.402686Z","iopub.status.idle":"2025-05-11T12:10:48.410513Z","shell.execute_reply.started":"2025-05-11T12:10:48.402657Z","shell.execute_reply":"2025-05-11T12:10:48.409829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"READ_TEST_PATH = \"/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/test.csv\"\nREAD_TRAIN_PATH = \"/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/train.csv\"\nprocessor = DataProcessing()\ntest_labels = processor.readData(READ_TEST_PATH)\ntrain_labels = processor.readData(READ_TRAIN_PATH)\ntrain_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-11T12:11:31.273194Z","iopub.execute_input":"2025-05-11T12:11:31.273500Z","iopub.status.idle":"2025-05-11T12:11:31.307244Z","shell.execute_reply.started":"2025-05-11T12:11:31.273477Z","shell.execute_reply":"2025-05-11T12:11:31.306465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_lis, test_lis = processor.dataLoader(test_labels, folder_path=\"/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/ot/ot\")\nlen(train_lis), len(test_lis)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-11T12:11:33.493025Z","iopub.execute_input":"2025-05-11T12:11:33.493325Z","iopub.status.idle":"2025-05-11T12:12:50.371615Z","shell.execute_reply.started":"2025-05-11T12:11:33.493302Z","shell.execute_reply":"2025-05-11T12:12:50.370879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_lis_pca = []\nprint(\"Applying PCA on training data ...\")\nfor img in train_lis:\n    res = processor.bandsPCA(img)\n    train_lis_pca.append(res)\n\n\ntest_lis_pca = []\nprint(\"Applying PCA on testing data ...\")\nfor img in test_lis:\n    res = processor.bandsPCA(img)\n    test_lis_pca.append(res)\nprint(\"PCA application is done for training and testing data ...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-11T12:12:50.372701Z","iopub.execute_input":"2025-05-11T12:12:50.373000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Collecting images of same shape for training purpose\ntrain_data = []\ntrain_labels_lis = []\nfor label, img in zip(train_labels['label'], train_lis_pca):\n    h, w, c = img.shape\n    if h == 128 and w == 128 and c == 3:\n        train_data.append(img)\n        train_labels_lis.append(label)\nlen(train_data), len(train_labels_lis)\n\ntest_data = []\nremoved_images = []\ntest_file_names = []\nfor img, file_name in zip(test_lis_pca, test_labels['id']):\n    h, w, c = img.shape\n    if h == 128 and w == 128 and c == 3:\n        test_data.append(img)\n        test_file_names.append(file_name)\n    else:\n        removed_images.append(file_name)\n        \nlen(test_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(test_lis_pca), len(test_labels), len(test_data), len(removed_images)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import mixed_precision\nmixed_precision.set_global_policy('mixed_float16') ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv3D, MaxPooling3D, Flatten, Dense, Dropout\nfrom tensorflow.keras.utils import to_categorical\nimport numpy as np\n\nX, y = np.array(train_data), np.array(train_labels_lis)\nnum_classes = 101\nepochs = 10\nbatch_size = 8\n\nmodel = Sequential([\n    Conv3D(32, kernel_size=(3, 3, 1), activation='relu', input_shape=(128, 128, 3, 1)),\n    MaxPooling3D(pool_size=(2, 2, 1)),\n\n    Conv3D(64, kernel_size=(3, 3, 1), activation='relu'),\n    MaxPooling3D(pool_size=(2, 2, 1)),\n\n    Conv3D(128, kernel_size=(3, 3, 1), activation='relu'),\n    MaxPooling3D(pool_size=(2, 2, 1)),\n\n    Flatten(),\n    Dense(256, activation='relu'),   \n    Dropout(0.5),\n    Dense(num_classes, activation='softmax')\n])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer='adam',\n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy'])\n\nX_train, y_train, X_test, y_test = np.array(X[:-200]), np.array(y[:-200]), np.array(X[-200:]), np.array(y[-200:])\n\nhistory = model.fit(X_train, y_train, \n                    validation_data = (X_test, y_test),\n                    epochs=epochs, \n                    batch_size=batch_size, \n                    )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report, recall_score, precision_score, f1_score\nX_test, y_test = X[-200:], y[-200:]\ny_pred = model.predict(X_test)\ny_pred = [np.argmax(val) for val in y_pred]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list1 = history.history['loss']  \nlist2 = history.history['val_loss']  \n\nepochs = range(1, len(list1) + 1)\n\nplt.figure(figsize=(8, 5))\nplt.plot(epochs, list1, label='Training loss', marker='o')\nplt.plot(epochs, list2, label='Validation loss\"', marker='s')\nplt.title('Model Accuracy Comparison')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-11T10:08:52.244918Z","iopub.execute_input":"2025-05-11T10:08:52.245208Z","iopub.status.idle":"2025-05-11T10:08:52.497489Z","shell.execute_reply.started":"2025-05-11T10:08:52.245187Z","shell.execute_reply":"2025-05-11T10:08:52.496801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history.history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-11T10:09:09.001853Z","iopub.execute_input":"2025-05-11T10:09:09.002426Z","iopub.status.idle":"2025-05-11T10:09:09.007372Z","shell.execute_reply.started":"2025-05-11T10:09:09.002402Z","shell.execute_reply":"2025-05-11T10:09:09.006727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list1 = history.history['accuracy']  \nlist2 = history.history['val_accuracy']  \n\nepochs = range(1, len(list1) + 1)\n\nplt.figure(figsize=(8, 5))\nplt.plot(epochs, list1, label='Training accuracy', marker='o')\nplt.plot(epochs, list2, label='Validation accuracy\"', marker='s')\nplt.title('Model Accuracy Comparison')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-11T10:09:10.521047Z","iopub.execute_input":"2025-05-11T10:09:10.521332Z","iopub.status.idle":"2025-05-11T10:09:10.719417Z","shell.execute_reply.started":"2025-05-11T10:09:10.521310Z","shell.execute_reply":"2025-05-11T10:09:10.718626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# processor.submission(test_df = test_labels, model = model, test_data = np.array(test_data), write_path='Submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-11T10:10:54.552892Z","iopub.execute_input":"2025-05-11T10:10:54.553561Z","iopub.status.idle":"2025-05-11T10:10:54.556473Z","shell.execute_reply.started":"2025-05-11T10:10:54.553535Z","shell.execute_reply":"2025-05-11T10:10:54.555892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}