{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":98450,"databundleVersionId":11749951,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-06T16:48:44.918955Z","iopub.execute_input":"2025-05-06T16:48:44.919581Z","iopub.status.idle":"2025-05-06T16:48:49.750736Z","shell.execute_reply.started":"2025-05-06T16:48:44.919543Z","shell.execute_reply":"2025-05-06T16:48:49.749685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Define dataset path\ndata_path = \"/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/ot/ot/\"\ntrain_csv_path = \"/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/train.csv\"\n\n# Load training data\ntrain_df = pd.read_csv(train_csv_path)\n\n# Create full paths for each npy file\ntrain_df[\"file_path\"] = train_df[\"id\"].apply(lambda x: os.path.join(data_path, x))\n\n# Check if paths are correctly assigned\nprint(train_df.head())  # Ensure file paths are linked properly","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T17:27:44.261679Z","iopub.execute_input":"2025-05-06T17:27:44.262372Z","iopub.status.idle":"2025-05-06T17:27:44.281221Z","shell.execute_reply.started":"2025-05-06T17:27:44.262344Z","shell.execute_reply":"2025-05-06T17:27:44.280289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load CSV file\ntrain_df = pd.read_csv(\"/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/train.csv\")\n\n# Print the column names\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T17:25:19.345909Z","iopub.execute_input":"2025-05-06T17:25:19.346189Z","iopub.status.idle":"2025-05-06T17:25:19.358339Z","shell.execute_reply.started":"2025-05-06T17:25:19.346167Z","shell.execute_reply":"2025-05-06T17:25:19.357316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T17:24:08.086477Z","iopub.execute_input":"2025-05-06T17:24:08.086889Z","iopub.status.idle":"2025-05-06T17:24:08.093215Z","shell.execute_reply.started":"2025-05-06T17:24:08.086865Z","shell.execute_reply":"2025-05-06T17:24:08.092153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nfor _, row in train_df.iterrows():\n    file_path = row[\"file_path\"]\n    \n    try:\n        img_data = np.load(file_path)\n        print(f\"File: {file_path}, Shape: {img_data.shape}\")\n    except Exception as e:\n        print(f\"Error loading {file_path}: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T17:30:49.816521Z","iopub.execute_input":"2025-05-06T17:30:49.816930Z","iopub.status.idle":"2025-05-06T17:31:40.841963Z","shell.execute_reply.started":"2025-05-06T17:30:49.816905Z","shell.execute_reply":"2025-05-06T17:31:40.840346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if img_data.shape != (128, 128, 125):\n    print(f\"Reshaping {file_path} from {img_data.shape} to (128, 128, 125)\")\n    img_data = np.resize(img_data, (128, 128, 125))  # Use cautiously!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T17:31:56.702293Z","iopub.execute_input":"2025-05-06T17:31:56.702708Z","iopub.status.idle":"2025-05-06T17:31:56.708917Z","shell.execute_reply.started":"2025-05-06T17:31:56.702616Z","shell.execute_reply":"2025-05-06T17:31:56.707861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if img_data.shape == (128, 128, 125):\n    X.append(img_data)\n    y.append(row[\"label\"])\nelse:\n    print(f\"Skipping {file_path} due to shape mismatch: {img_data.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T17:32:31.836098Z","iopub.execute_input":"2025-05-06T17:32:31.836415Z","iopub.status.idle":"2025-05-06T17:32:31.842632Z","shell.execute_reply.started":"2025-05-06T17:32:31.836394Z","shell.execute_reply":"2025-05-06T17:32:31.841435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = np.array(X)\ny = np.array(y)\nprint(f\"Final dataset shape: {X.shape}, Labels shape: {y.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T17:32:51.611729Z","iopub.execute_input":"2025-05-06T17:32:51.612082Z","iopub.status.idle":"2025-05-06T17:32:51.629053Z","shell.execute_reply.started":"2025-05-06T17:32:51.612059Z","shell.execute_reply":"2025-05-06T17:32:51.627821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nX, y = [], []\n\n# Loop through dataset to load images\nfor _, row in train_df.iterrows():\n    file_path = row[\"file_path\"]\n    label = row[\"label\"]\n\n    # Load the image file\n    img_data = np.load(file_path)\n\n    X.append(img_data)  # Store image data\n    y.append(label)  # Store corresponding label\n\n# Convert to NumPy arrays\nX = np.array(X)\ny = np.array(y)\n\nprint(f\"Loaded {len(X)} samples with shape {X[0].shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T17:29:03.396067Z","iopub.execute_input":"2025-05-06T17:29:03.396710Z","iopub.status.idle":"2025-05-06T17:29:42.515203Z","shell.execute_reply.started":"2025-05-06T17:29:03.396683Z","shell.execute_reply":"2025-05-06T17:29:42.513970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}