{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":98450,"databundleVersionId":11749951,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Using Autogluon for this challenge\n\nIn this notebook I use the MultiModalPredictor function from Autogluon to predict the labels.\n\nAutogluon can handle images, so let's see how does it fare with this challenge. ","metadata":{}},{"cell_type":"code","source":"# Install necessary libraries\n!pip install -q autogluon","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:38:43.242312Z","iopub.execute_input":"2025-05-12T10:38:43.242548Z","iopub.status.idle":"2025-05-12T10:40:51.799223Z","shell.execute_reply.started":"2025-05-12T10:38:43.242527Z","shell.execute_reply":"2025-05-12T10:40:51.798133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\ndef is_interactive():\n   return os.environ.get('KAGGLE_KERNEL_RUN_TYPE','') == \"Interactive\"\nprint(\"is interactive session?\", is_interactive())\npreset_quality = \"medium_quality\" if is_interactive() else \"best_quality\"\n\ntime_limit = 60 if is_interactive() else 3600","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:40:51.801727Z","iopub.execute_input":"2025-05-12T10:40:51.802011Z","iopub.status.idle":"2025-05-12T10:40:51.808510Z","shell.execute_reply.started":"2025-05-12T10:40:51.801985Z","shell.execute_reply":"2025-05-12T10:40:51.807688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import autogluon.core as ag\n#from autogluon import ImagePredictor\nimport pandas as pd\nimport os\nimport numpy as np\nfrom autogluon.multimodal import MultiModalPredictor\nfrom PIL import Image\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:40:51.809573Z","iopub.execute_input":"2025-05-12T10:40:51.809866Z","iopub.status.idle":"2025-05-12T10:41:26.676832Z","shell.execute_reply.started":"2025-05-12T10:40:51.809836Z","shell.execute_reply":"2025-05-12T10:41:26.676083Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Convert images to PNG\n\nAutogluon doesn't read .npy images, so we save them as png.\n\nSome images seem to be corrupted, or have a wrong shape. We'll skip them this time.","metadata":{}},{"cell_type":"code","source":"\n# Read the train.csv file that contains the labels\ntrain_df = pd.read_csv('/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/train.csv')\n\nprint(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:41:26.677577Z","iopub.execute_input":"2025-05-12T10:41:26.678473Z","iopub.status.idle":"2025-05-12T10:41:26.712993Z","shell.execute_reply.started":"2025-05-12T10:41:26.678448Z","shell.execute_reply":"2025-05-12T10:41:26.712096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_folder = \"/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/ot/ot\"\n# Define the output directory where you will save the images\noutput_dir = '/kaggle/working/hyperspectral_images/'\n\n# Create the directory if it doesn't exist\nos.makedirs(output_dir, exist_ok=True)\n\n# Function to load .npy files\ndef load_npy_image(image_name,\n                  image_folder = \"/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/ot/ot\"):\n    image_path = os.path.join(image_folder, image_name)\n    image = np.load(image_path)\n    return image\n\n# Convert the .npy files to image files and save them\ndef save_image_from_npy(image_name, image_data,\n                       output_dir = '/kaggle/working/hyperspectral_images/'\n                       ):\n    # Ensure image has correct dimensions (e.g., 128x128x125)\n    if image_data.shape == (128, 128, 125):\n        # Normalize the data to the range 0-255\n        # Here, we'll just take the first band for simplicity\n        image = image_data[:, :, 0]  # Extract the first band (change if you want to visualize different bands)\n        image = np.clip(image, 0, 255)  # Clip values to valid image range\n        image = image.astype(np.uint8)  # Convert to unsigned 8-bit integer type\n        image_path = os.path.join(output_dir, image_name.replace('.npy', '.png'))\n        # Convert numpy array to image using PIL\n        pil_image = Image.fromarray(image)\n        pil_image.save(image_path)\n        #print(f\"Saved {image_path}\")\n    else:\n        print(f\"Skipping {image_name} due to unexpected shape {image_data.shape}\")\n\n# Loop through the training data and save the images\nfor idx, row in train_df.iterrows():\n    try:\n        # Load the .npy image\n        image = load_npy_image(row['id'])\n        \n        # Check the shape of the image and reshape if necessary\n        save_image_from_npy(row['id'], image)\n    \n    except ValueError as e:\n        print(f\"Error loading {row['id']}: {e}\")\n        continue  # Skip the image and continue with the next one\n        \n# Map the image filenames to the new .png file paths\ntrain_df['image_path'] = train_df['id'].apply(lambda x: os.path.join(output_dir, x.replace('.npy', '.png')))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:41:26.715116Z","iopub.execute_input":"2025-05-12T10:41:26.715454Z","iopub.status.idle":"2025-05-12T10:42:33.476909Z","shell.execute_reply.started":"2025-05-12T10:41:26.715430Z","shell.execute_reply":"2025-05-12T10:42:33.476074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA of this mysterious dataset\n\nI can't make head or tails of this - some images are blank, yet they have high labels.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef plot_image(train_df, row_id):\n    img = Image.open(train_df.iloc[row_id,2])\n    plt.imshow(img)\n    plt.title(f\"{train_df.iloc[row_id,0]} - {train_df.iloc[row_id,1]}\")\nplot_image(train_df, 6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:53:13.088945Z","iopub.execute_input":"2025-05-12T10:53:13.089360Z","iopub.status.idle":"2025-05-12T10:53:13.341292Z","shell.execute_reply.started":"2025-05-12T10:53:13.089337Z","shell.execute_reply":"2025-05-12T10:53:13.340196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\n\n# Define a function to plot images in a grid\ndef plot_images_in_grid(train_df, row_ids, grid_size=(10, 7)):\n    # Create a figure with a specified size\n    fig, axes = plt.subplots(nrows=grid_size[0], ncols=grid_size[1], figsize=(20, 14))\n    \n    # Flatten the axes array for easier iteration\n    axes = axes.flatten()\n    \n    # Loop through the row_ids and plot images\n    for i, row_id in enumerate(row_ids):\n        img = Image.open(train_df.iloc[row_id, 2])  # Load the image from the path\n        axes[i].imshow(img)  # Display the image\n        axes[i].set_title(f\"{train_df.iloc[row_id, 0]} - {train_df.iloc[row_id, 1]}\")\n        axes[i].axis('off')  # Turn off the axis to keep the image clean\n    \n    # Adjust the layout to prevent overlapping titles and images\n    plt.tight_layout()\n    plt.show()\n\n# Example usage:\n# Choose the first 70 row IDs for plotting\nrow_ids = list(range(70))\n\n# Call the function to plot the images\nplot_images_in_grid(train_df, row_ids, grid_size=(10, 7))  # Adjust grid_size as needed\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:53:19.034209Z","iopub.execute_input":"2025-05-12T10:53:19.034541Z","iopub.status.idle":"2025-05-12T10:53:23.806518Z","shell.execute_reply.started":"2025-05-12T10:53:19.034517Z","shell.execute_reply":"2025-05-12T10:53:23.804995Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Now the dataframe has 'image_path' and 'label' columns\n# Preview the updated dataframe\nprint(train_df.head())\n\ntrain_df.iloc[0,2]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:42:33.788854Z","iopub.execute_input":"2025-05-12T10:42:33.789184Z","iopub.status.idle":"2025-05-12T10:42:33.798442Z","shell.execute_reply.started":"2025-05-12T10:42:33.789161Z","shell.execute_reply":"2025-05-12T10:42:33.797467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/working/hyperspectral_images/sample697.png","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:42:33.799516Z","iopub.execute_input":"2025-05-12T10:42:33.800798Z","iopub.status.idle":"2025-05-12T10:42:33.987646Z","shell.execute_reply.started":"2025-05-12T10:42:33.800616Z","shell.execute_reply":"2025-05-12T10:42:33.986144Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Running Autogluon prediction","metadata":{}},{"cell_type":"code","source":"from autogluon.multimodal import MultiModalPredictor\n#from autogluon.vision import ImagePredictor\n\n# Initialize AutoGluon MultiModalPredictor with label column name\npredictor = MultiModalPredictor(label=\"label\")\n\n# Perform k-fold cross-validation (e.g., 5-fold)\ncv_results = predictor.fit(\n    train_data=train_df.drop(columns=\"id\"),\n    time_limit=time_limit,\n#    num_bagging_folds=5,  # Number of folds for cross-validation\n#    num_bagging_sets=1,   # Number of models to be trained per fold\n    save_space=True,\n    presets=preset_quality\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:42:33.989147Z","iopub.execute_input":"2025-05-12T10:42:33.989481Z","iopub.status.idle":"2025-05-12T10:43:45.788108Z","shell.execute_reply.started":"2025-05-12T10:42:33.989436Z","shell.execute_reply":"2025-05-12T10:43:45.787128Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Plotting CV scores","metadata":{}},{"cell_type":"code","source":"cv_results.fit_summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:43:45.789588Z","iopub.execute_input":"2025-05-12T10:43:45.789997Z","iopub.status.idle":"2025-05-12T10:43:45.797063Z","shell.execute_reply.started":"2025-05-12T10:43:45.789956Z","shell.execute_reply":"2025-05-12T10:43:45.796267Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Predicting and creating submission","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/test.csv')\n\n# Loop through the training data and save the images\nfor idx, row in test_df.iterrows():\n    try:\n        # Load the .npy image\n        image = load_npy_image(row['id'])\n        \n        # Check the shape of the image and reshape if necessary\n        save_image_from_npy(row['id'], image)\n    \n    except ValueError as e:\n        print(f\"Error loading {row['id']}: {e}\")\n        continue  # Skip the image and continue with the next one\n\n\ntest_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:43:45.797961Z","iopub.execute_input":"2025-05-12T10:43:45.798324Z","iopub.status.idle":"2025-05-12T10:44:04.619370Z","shell.execute_reply.started":"2025-05-12T10:43:45.798293Z","shell.execute_reply":"2025-05-12T10:44:04.618447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"      \n# Map the image filenames to the new .png file paths\ntest_df['image_path'] = test_df['id'].apply(lambda x: os.path.join(output_dir, x.replace('.npy', '.png')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:44:04.620296Z","iopub.execute_input":"2025-05-12T10:44:04.620546Z","iopub.status.idle":"2025-05-12T10:44:04.627531Z","shell.execute_reply.started":"2025-05-12T10:44:04.620520Z","shell.execute_reply":"2025-05-12T10:44:04.626500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:44:04.628479Z","iopub.execute_input":"2025-05-12T10:44:04.628771Z","iopub.status.idle":"2025-05-12T10:44:04.651218Z","shell.execute_reply.started":"2025-05-12T10:44:04.628741Z","shell.execute_reply":"2025-05-12T10:44:04.650339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/working/hyperspectral_images/sample1957.*","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:44:04.652315Z","iopub.execute_input":"2025-05-12T10:44:04.652599Z","iopub.status.idle":"2025-05-12T10:44:04.851674Z","shell.execute_reply.started":"2025-05-12T10:44:04.652578Z","shell.execute_reply":"2025-05-12T10:44:04.850641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions using the trained AutoGluon model\n# Ensure we're passing the entire DataFrame, not just the image paths column\npredictions = predictor.predict(test_df[['image_path']])\n\n# Prepare the submission DataFrame\nsubmission_df = pd.DataFrame({\n    'ID': test_df['id'],  # The IDs from the test.csv\n    'label': predictions   # The predicted labels from the model\n})\n\n# Save the submission file\nsubmission_df.to_csv('submission.csv', index=False)\n\n# Optionally, show the first few rows of the submission file\nprint(submission_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:44:04.852927Z","iopub.execute_input":"2025-05-12T10:44:04.853228Z","iopub.status.idle":"2025-05-12T10:44:18.145527Z","shell.execute_reply.started":"2025-05-12T10:44:04.853196Z","shell.execute_reply":"2025-05-12T10:44:18.144479Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualizing predictions\n\nThe distribution of predicted labels seems very different compared to the train, so I don't expect this prediction to have a great leaderboard score.","metadata":{}},{"cell_type":"code","source":"!ls ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:44:18.146840Z","iopub.execute_input":"2025-05-12T10:44:18.147366Z","iopub.status.idle":"2025-05-12T10:44:18.332717Z","shell.execute_reply.started":"2025-05-12T10:44:18.147322Z","shell.execute_reply":"2025-05-12T10:44:18.331374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nsns.histplot(data=train_df, x=\"label\").set_title(\"distribution of labels in Train\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:44:18.334179Z","iopub.execute_input":"2025-05-12T10:44:18.334534Z","iopub.status.idle":"2025-05-12T10:44:19.820529Z","shell.execute_reply.started":"2025-05-12T10:44:18.334493Z","shell.execute_reply":"2025-05-12T10:44:19.819623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nsns.histplot(data=submission_df, x=\"label\").set_title(\"distribution of predicted labels\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:44:19.821653Z","iopub.execute_input":"2025-05-12T10:44:19.822494Z","iopub.status.idle":"2025-05-12T10:44:20.059339Z","shell.execute_reply.started":"2025-05-12T10:44:19.822471Z","shell.execute_reply":"2025-05-12T10:44:20.058397Z"}},"outputs":[],"execution_count":null}]}