{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":98450,"databundleVersionId":11749951,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install umap-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:45:19.547493Z","iopub.execute_input":"2025-04-29T09:45:19.547816Z","iopub.status.idle":"2025-04-29T09:45:27.663604Z","shell.execute_reply.started":"2025-04-29T09:45:19.547786Z","shell.execute_reply":"2025-04-29T09:45:27.662214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.decomposition import PCA\nimport umap.umap_ as umap\nimport os\nfrom scipy.cluster.hierarchy import dendrogram, linkage\nfrom scipy.stats import f_oneway, kruskal\nfrom scipy.stats import levene, bartlett\nfrom scipy.stats import ttest_ind, mannwhitneyu\nfrom scipy.stats import shapiro\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:58:02.768591Z","iopub.execute_input":"2025-04-29T09:58:02.768911Z","iopub.status.idle":"2025-04-29T09:58:02.775267Z","shell.execute_reply.started":"2025-04-29T09:58:02.768888Z","shell.execute_reply":"2025-04-29T09:58:02.774235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/ot/ot'\nCSV_PATH = '/kaggle/input/beyond-visible-spectrum-ai-for-agriculture-2025/train.csv'\n\ndf = pd.read_csv(CSV_PATH)\n\ndata_list = []\nlabels = []\n\nexpected_shape = (128, 128, 125) \n\nfor _, row in df.iterrows():\n    file_path = os.path.join(DATA_DIR, row['id'])\n    try:\n        cube = np.load(file_path)\n\n        if cube.shape != expected_shape:\n            continue  \n\n        mean_spectrum = cube.reshape(-1, cube.shape[2]).mean(axis=0)\n        data_list.append(mean_spectrum)\n        labels.append(row['label'])\n\n    except Exception as e:\n        print(f\"Error with {file_path}: {e}\")\n\nX = np.array(data_list)\ny = np.array(labels)\n\nreducer = umap.UMAP(random_state=42)\nX_embedded = reducer.fit_transform(X)\n\nplt.figure(figsize=(10, 8), constrained_layout=True)\nsns.scatterplot(x=X_embedded[:, 0], y=X_embedded[:, 1], hue=y, palette='tab10')\nplt.title('UMAP Projection of Hyperspectral Data by Label')\nplt.xlabel('UMAP 1')\nplt.ylabel('UMAP 2')\nplt.legend(title='Label', bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.savefig(\"umap_plot.png\")\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:47:18.977972Z","iopub.execute_input":"2025-04-29T09:47:18.978368Z","iopub.status.idle":"2025-04-29T09:48:41.580040Z","shell.execute_reply.started":"2025-04-29T09:47:18.978341Z","shell.execute_reply":"2025-04-29T09:48:41.578841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_spectra = pd.DataFrame(X)\ndf_spectra['id'] = y\n\nplt.figure(figsize=(12, 6))\nfor label in df_spectra['id'].unique():\n    mean_spectrum = df_spectra[df_spectra['id'] == label].drop('id', axis=1).mean()\n    plt.plot(mean_spectrum, label=label)\n\nplt.title('Mean Reflectance Spectra per Class')\nplt.xlabel('Bands (Spectral Channels)')\nplt.ylabel('Reflectance')\nplt.legend(title='Class Label')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:49:03.705750Z","iopub.execute_input":"2025-04-29T09:49:03.706140Z","iopub.status.idle":"2025-04-29T09:49:05.167014Z","shell.execute_reply.started":"2025-04-29T09:49:03.706090Z","shell.execute_reply":"2025-04-29T09:49:05.166098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_plot = df_spectra.copy()\nselected_bands = [10, 50, 100] \n\nfor band in selected_bands:\n    plt.figure(figsize=(10, 5))\n    sns.boxplot(data=df_plot, x='id', y=band)\n    plt.title(f'Boxplot for Band {band}')\n    plt.xlabel('Class')\n    plt.ylabel(f'Reflectance at Band {band}')\n    plt.tight_layout()\n    plt.show()\n\n    plt.figure(figsize=(10, 5))\n    sns.violinplot(data=df_plot, x='id', y=band)\n    plt.title(f'Violinplot for Band {band}')\n    plt.xlabel('Class')\n    plt.ylabel(f'Reflectance at Band {band}')\n    plt.tight_layout()\n    plt.show","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:49:39.829787Z","iopub.execute_input":"2025-04-29T09:49:39.830216Z","iopub.status.idle":"2025-04-29T09:49:52.443277Z","shell.execute_reply.started":"2025-04-29T09:49:39.830184Z","shell.execute_reply":"2025-04-29T09:49:52.442396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlation_matrix = df_spectra.drop('id', axis=1).corr()\n\ncorr_unstacked = correlation_matrix.where(np.triu(np.ones(correlation_matrix.shape), k=1).astype(bool))\ncorr_pairs = corr_unstacked.unstack().dropna()\ntop_corr = corr_pairs.abs().sort_values(ascending=False).head(10)\n\nprint(\"Top 10 most correlated band pairs (by absolute correlation):\")\nfor (band1, band2), corr_val in top_corr.items():\n    print(f\"Bands {band1} & {band2}: correlation = {corr_val:.3f}\")\n\nplt.figure(figsize=(12, 10))\nsns.heatmap(correlation_matrix, cmap='coolwarm', center=0, square=True)\nplt.title('Spectral Band Correlation Heatmap')\nplt.xlabel('Band')\nplt.ylabel('Band')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:52:21.448722Z","iopub.execute_input":"2025-04-29T09:52:21.449248Z","iopub.status.idle":"2025-04-29T09:52:22.517471Z","shell.execute_reply.started":"2025-04-29T09:52:21.449217Z","shell.execute_reply":"2025-04-29T09:52:22.516283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mean_spectra_by_class = df_spectra.groupby('id').mean()\nlinked = linkage(mean_spectra_by_class, method='ward')\n\nplt.figure(figsize=(10, 6))\ndendrogram(linked, labels=mean_spectra_by_class.index.tolist(), leaf_rotation=90)\nplt.title('Dendrogram of Class Mean Spectra')\nplt.xlabel('Class')\nplt.ylabel('Distance')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:51:13.901581Z","iopub.execute_input":"2025-04-29T09:51:13.901917Z","iopub.status.idle":"2025-04-29T09:51:15.080886Z","shell.execute_reply.started":"2025-04-29T09:51:13.901892Z","shell.execute_reply":"2025-04-29T09:51:15.079662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"R_band, G_band, B_band = 90, 60, 30\n\ndef create_rgb_image_for_class(class_label):\n    sample_path = os.path.join(DATA_DIR, df[df['label'] == class_label]['id'].iloc[0])\n    cube = np.load(sample_path)\n\n    rgb_image = np.stack([\n        cube[:, :, R_band],\n        cube[:, :, G_band],\n        cube[:, :, B_band]\n    ], axis=-1)\n\n    \n    rgb_image = (rgb_image - rgb_image.min()) / (rgb_image.max() - rgb_image.min())\n    return rgb_image\n\n\nclass_labels = df['label'].unique()[:4]  \nfig, axes = plt.subplots(2, 2, figsize=(12, 12))\n\nfor ax, class_label in zip(axes.flatten(), class_labels):\n    rgb_image = create_rgb_image_for_class(class_label)\n    ax.imshow(rgb_image)\n    ax.set_title(f'Class: {class_label}')\n    ax.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:53:47.547459Z","iopub.execute_input":"2025-04-29T09:53:47.547798Z","iopub.status.idle":"2025-04-29T09:53:48.111155Z","shell.execute_reply.started":"2025-04-29T09:53:47.547775Z","shell.execute_reply":"2025-04-29T09:53:48.110033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"anova_results = {}\nfor band in range(X.shape[1]): \n    groups = [X[y == label, band] for label in np.unique(y)]  \n    f_stat, p_value = f_oneway(*groups)\n    anova_results[band] = p_value\n\n\nsignificant_bands_anova = {band: p for band, p in anova_results.items() if p < 0.05}\nprint(\"ANOVA significant bands:\", significant_bands_anova)\n\nkruskal_results = {}\nfor band in range(X.shape[1]):\n    groups = [X[y == label, band] for label in np.unique(y)] \n    h_stat, p_value = kruskal(*groups)\n    kruskal_results[band] = p_value\n\nsignificant_bands_kruskal = {band: p for band, p in kruskal_results.items() if p < 0.05}\nprint(\"Kruskal-Wallis significant bands:\", significant_bands_kruskal)\n\nplt.figure(figsize=(8, 6))\nsns.countplot(x=y)\nplt.title(\"Class Distribution\")\nplt.xlabel(\"Class\")\nplt.ylabel(\"Frequency\")\nplt.show()\n\nfor band in range(X.shape[1]):\n    _, p_value = shapiro(X[:, band])\n    print(f\"Shapiro-Wilk test for Band {band}: p-value = {p_value}\")\n\npca = PCA(n_components=2)\nX_pca = pca.fit_transform(X)\n\nplt.figure(figsize=(10, 8))\nsns.scatterplot(x=X_pca[:, 0], y=X_pca[:, 1], hue=y, palette='tab10')\nplt.title('PCA Projection')\nplt.xlabel('PCA 1')\nplt.ylabel('PCA 2')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T09:58:50.447690Z","iopub.execute_input":"2025-04-29T09:58:50.448027Z","iopub.status.idle":"2025-04-29T09:58:55.715963Z","shell.execute_reply.started":"2025-04-29T09:58:50.448007Z","shell.execute_reply":"2025-04-29T09:58:55.715131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class1_data = X[y == df['label'].unique()[0]]  \nclass2_data = X[y == df['label'].unique()[1]] \n\n\nt_stat, p_value_t = ttest_ind(class1_data, class2_data, axis=0)\n\nu_stat, p_value_u = mannwhitneyu(class1_data.flatten(), class2_data.flatten())\n\nprint(\"t-test p-values:\", p_value_t)\nprint(\"Mann-Whitney U test p-value:\", p_value_u)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T10:01:51.217022Z","iopub.execute_input":"2025-04-29T10:01:51.218227Z","iopub.status.idle":"2025-04-29T10:01:51.237111Z","shell.execute_reply.started":"2025-04-29T10:01:51.218181Z","shell.execute_reply":"2025-04-29T10:01:51.236273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"levene_results = {}\nfor band in range(X.shape[1]):\n    groups = [X[y == label, band] for label in np.unique(y)]\n    stat, p_value = levene(*groups)\n    levene_results[band] = p_value\n\nbartlett_results = {}\nfor band in range(X.shape[1]):\n    groups = [X[y == label, band] for label in np.unique(y)]\n    stat, p_value = bartlett(*groups)\n    bartlett_results[band] = p_value\n\nsignificant_bands_levene = {band: p for band, p in levene_results.items() if p < 0.05}\nsignificant_bands_bartlett = {band: p for band, p in bartlett_results.items() if p < 0.05}\n\nprint(\"Levene Test significant bands:\", significant_bands_levene)\nprint(\"Bartlett Test significant bands:\", significant_bands_bartlett)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T10:05:04.694994Z","iopub.execute_input":"2025-04-29T10:05:04.695526Z","iopub.status.idle":"2025-04-29T10:05:07.395260Z","shell.execute_reply.started":"2025-04-29T10:05:04.695489Z","shell.execute_reply":"2025-04-29T10:05:07.394283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_ndvi(cube, red_band=30, nir_band=90):\n    red = cube[:, :, red_band]\n    nir = cube[:, :, nir_band]\n    return (nir - red) / (nir + red)\n\nsample_path = os.path.join(DATA_DIR, df[df['label'] == df['label'].unique()[0]]['id'].iloc[0])\ncube = np.load(sample_path)\nndvi_image = calculate_ndvi(cube)\n\nplt.imshow(ndvi_image, cmap='RdYlGn')\nplt.title(\"NDVI (NIR-Red)/(NIR+Red)\")\nplt.colorbar()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T10:05:35.651467Z","iopub.execute_input":"2025-04-29T10:05:35.652000Z","iopub.status.idle":"2025-04-29T10:05:35.972875Z","shell.execute_reply.started":"2025-04-29T10:05:35.651964Z","shell.execute_reply":"2025-04-29T10:05:35.972027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.cluster import KMeans\n\nkmeans = KMeans(n_clusters=25, random_state=42)\nkmeans.fit(X)\n\nplt.scatter(X_embedded[:, 0], X_embedded[:, 1], c=kmeans.labels_, cmap='viridis')\nplt.title(\"KMeans Clustering\")\nplt.xlabel('UMAP 1')\nplt.ylabel('UMAP 2')\nplt.colorbar(label='Cluster')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T10:06:09.687897Z","iopub.execute_input":"2025-04-29T10:06:09.688328Z","iopub.status.idle":"2025-04-29T10:06:10.286754Z","shell.execute_reply.started":"2025-04-29T10:06:09.688298Z","shell.execute_reply":"2025-04-29T10:06:10.285522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_snr(X, y, class_label):\n    class_data = X[y == class_label]\n    mean_signal = class_data.mean(axis=0)\n    noise = class_data.std(axis=0)\n    snr = mean_signal / noise\n    return snr\n\nsnr_values = {}\nfor label in np.unique(y):\n    snr_values[label] = calculate_snr(X, y, label)\n\nfor label, snr in snr_values.items():\n    print(f\"SNR for class {label}: {snr[:5]}\")  \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T10:06:28.594335Z","iopub.execute_input":"2025-04-29T10:06:28.594673Z","iopub.status.idle":"2025-04-29T10:06:28.622006Z","shell.execute_reply.started":"2025-04-29T10:06:28.594651Z","shell.execute_reply":"2025-04-29T10:06:28.621042Z"}},"outputs":[],"execution_count":null}]}