{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81000,"databundleVersionId":8812083,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:17:31.579521Z","iopub.execute_input":"2025-04-15T03:17:31.579825Z","iopub.status.idle":"2025-04-15T03:17:32.33917Z","shell.execute_reply.started":"2025-04-15T03:17:31.579801Z","shell.execute_reply":"2025-04-15T03:17:32.338209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data manipulation\nimport pandas as pd\nimport numpy as np\n\n# Data visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Preprocessing\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\n\n# Models\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# Feature selection\nfrom sklearn.decomposition import PCA\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:25:05.147993Z","iopub.execute_input":"2025-04-15T03:25:05.148504Z","iopub.status.idle":"2025-04-15T03:25:06.787949Z","shell.execute_reply.started":"2025-04-15T03:25:05.148477Z","shell.execute_reply":"2025-04-15T03:25:06.787135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Maize climate feature files\npr_maize = pd.read_parquet('/kaggle/input/the-future-crop-challenge/pr_maize_train.parquet')\ntas_maize = pd.read_parquet('/kaggle/input/the-future-crop-challenge/tas_maize_train.parquet')\ntasmax_maize = pd.read_parquet('/kaggle/input/the-future-crop-challenge/tasmax_maize_train.parquet')\ntasmin_maize = pd.read_parquet('/kaggle/input/the-future-crop-challenge/tasmin_maize_train.parquet')\nrsds_maize = pd.read_parquet('/kaggle/input/the-future-crop-challenge/rsds_maize_train.parquet')\nsoil_co2_maize = pd.read_parquet('/kaggle/input/the-future-crop-challenge/soil_co2_maize_train.parquet')\n\n# Target variable\nmaize_yield = pd.read_parquet('/kaggle/input/the-future-crop-challenge/train_solutions_maize.parquet')\n\n# Preview\nprint(\"PR Maize shape:\", pr_maize.shape)\nprint(\"TAS Maize shape:\", tas_maize.shape)\nprint(\"TASMAX Maize shape:\", tasmax_maize.shape)\nprint(\"TASMIN Maize shape:\", tasmin_maize.shape)\nprint(\"RSDS Maize shape:\", rsds_maize.shape)\nprint(\"Soil CO2 Maize shape:\", soil_co2_maize.shape)\nprint(\"Target Yield shape:\", maize_yield.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:25:19.841745Z","iopub.execute_input":"2025-04-15T03:25:19.842195Z","iopub.status.idle":"2025-04-15T03:25:49.372048Z","shell.execute_reply.started":"2025-04-15T03:25:19.842166Z","shell.execute_reply":"2025-04-15T03:25:49.371161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check missing values for each dataframe\nprint(\"Missing in pr_maize:\", pr_maize.isnull().sum().sum())\nprint(\"Missing in tas_maize:\", tas_maize.isnull().sum().sum())\nprint(\"Missing in tasmax_maize:\", tasmax_maize.isnull().sum().sum())\nprint(\"Missing in tasmin_maize:\", tasmin_maize.isnull().sum().sum())\nprint(\"Missing in rsds_maize:\", rsds_maize.isnull().sum().sum())\nprint(\"Missing in soil_co2_maize:\", soil_co2_maize.isnull().sum().sum())\nprint(\"Missing in maize_yield:\", maize_yield.isnull().sum().sum())\n\n# Also look at the columns of each\nprint(\"\\nSoil CO2 Columns:\")\nprint(soil_co2_maize.columns)\nprint(\"\\nYield Preview:\")\nprint(maize_yield.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:27:48.092326Z","iopub.execute_input":"2025-04-15T03:27:48.092989Z","iopub.status.idle":"2025-04-15T03:27:49.810186Z","shell.execute_reply.started":"2025-04-15T03:27:48.092962Z","shell.execute_reply":"2025-04-15T03:27:49.809222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to check non-numeric columns\ndef check_non_numeric(df, name):\n    non_numeric = df.select_dtypes(exclude='number').columns\n    print(f\"❗ Non-numeric columns in {name}: {list(non_numeric)}\")\n\ncheck_non_numeric(pr_maize, \"pr_maize\")\ncheck_non_numeric(tas_maize, \"tas_maize\")\ncheck_non_numeric(tasmax_maize, \"tasmax_maize\")\ncheck_non_numeric(tasmin_maize, \"tasmin_maize\")\ncheck_non_numeric(rsds_maize, \"rsds_maize\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:30:37.145898Z","iopub.execute_input":"2025-04-15T03:30:37.146461Z","iopub.status.idle":"2025-04-15T03:30:37.177276Z","shell.execute_reply.started":"2025-04-15T03:30:37.146409Z","shell.execute_reply":"2025-04-15T03:30:37.1764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove non-numeric columns and compute stats\nclimate_features = pd.DataFrame()\n\nfor df, name in zip([pr_maize, tas_maize, tasmax_maize, tasmin_maize, rsds_maize],\n                    [\"pr\", \"tas\", \"tasmax\", \"tasmin\", \"rsds\"]):\n    \n    # Remove non-numeric columns\n    df_clean = df.drop(columns=['crop', 'variable'])\n    \n    # Compute row-wise mean and std\n    climate_features[f\"{name}_mean\"] = df_clean.mean(axis=1)\n    climate_features[f\"{name}_std\"] = df_clean.std(axis=1)\n\nclimate_features.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:31:19.565452Z","iopub.execute_input":"2025-04-15T03:31:19.565758Z","iopub.status.idle":"2025-04-15T03:31:26.117424Z","shell.execute_reply.started":"2025-04-15T03:31:19.565734Z","shell.execute_reply":"2025-04-15T03:31:26.116576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Keep only numeric columns from soil_co2_maize\nsoil_features = soil_co2_maize[['co2', 'nitrogen']].copy()\n\n# Reset index to make sure alignment works\nsoil_features.reset_index(drop=True, inplace=True)\nclimate_features.reset_index(drop=True, inplace=True)\n\n# Merge climate and soil features\nfinal_features = pd.concat([climate_features, soil_features], axis=1)\nfinal_features.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:34:03.356997Z","iopub.execute_input":"2025-04-15T03:34:03.357806Z","iopub.status.idle":"2025-04-15T03:34:03.416928Z","shell.execute_reply.started":"2025-04-15T03:34:03.357779Z","shell.execute_reply":"2025-04-15T03:34:03.416034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reset index to ensure alignment\nmaize_yield.reset_index(drop=True, inplace=True)\n\n# Combine features and target\nmaize_data = pd.concat([final_features, maize_yield], axis=1)\n\n# Preview\nmaize_data.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:40:48.621394Z","iopub.execute_input":"2025-04-15T03:40:48.62172Z","iopub.status.idle":"2025-04-15T03:40:48.675213Z","shell.execute_reply.started":"2025-04-15T03:40:48.621698Z","shell.execute_reply":"2025-04-15T03:40:48.674255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Separate features and target\nX = maize_data.drop('yield', axis=1)\ny = maize_data['yield']\n\n# Standardize features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:41:42.696295Z","iopub.execute_input":"2025-04-15T03:41:42.697134Z","iopub.status.idle":"2025-04-15T03:41:42.832247Z","shell.execute_reply.started":"2025-04-15T03:41:42.6971Z","shell.execute_reply":"2025-04-15T03:41:42.831316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# PCA to retain 95% variance\npca = PCA(n_components=0.95)\nX_pca = pca.fit_transform(X_scaled)\n\n# Print how many components were kept\nprint(f\"Selected {pca.n_components_} principal components.\")\n\n# Optional: Scree plot\nplt.figure(figsize=(8,4))\nplt.plot(np.cumsum(pca.explained_variance_ratio_), marker='o')\nplt.xlabel('Number of Components')\nplt.ylabel('Cumulative Explained Variance')\nplt.title('PCA Scree Plot')\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:41:52.07256Z","iopub.execute_input":"2025-04-15T03:41:52.072877Z","iopub.status.idle":"2025-04-15T03:41:52.586919Z","shell.execute_reply.started":"2025-04-15T03:41:52.072852Z","shell.execute_reply":"2025-04-15T03:41:52.586027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming you have already loaded the maize data into individual DataFrames like:\n# pr_maize, tas_maize, tasmax_maize, tasmin_maize, rsds_maize\n\n# Clean the data and compute mean and std for each feature\nclimate_features = pd.DataFrame()\n\nfor df, name in zip([pr_maize, tas_maize, tasmax_maize, tasmin_maize, rsds_maize],\n                    [\"pr\", \"tas\", \"tasmax\", \"tasmin\", \"rsds\"]):\n    \n    # Remove non-numeric columns\n    df_clean = df.drop(columns=['crop', 'variable'], errors='ignore')  # 'errors' to ignore missing columns\n    \n    # Compute row-wise mean and std\n    climate_features[f\"{name}_mean\"] = df_clean.mean(axis=1)\n    climate_features[f\"{name}_std\"] = df_clean.std(axis=1)\n\n# Assuming you want to concatenate these features into a final DataFrame\nfinal_df = climate_features\n\n# Summary statistics\nsummary_stats = final_df.describe()\nprint(summary_stats)\n\n# Distribution plots for all features\nfinal_df.hist(bins=30, figsize=(18, 12), layout=(5, 3))\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:48:02.115497Z","iopub.execute_input":"2025-04-15T03:48:02.11627Z","iopub.status.idle":"2025-04-15T03:48:12.208065Z","shell.execute_reply.started":"2025-04-15T03:48:02.116235Z","shell.execute_reply":"2025-04-15T03:48:12.207046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.feature_selection import VarianceThreshold\n\n# Assuming final_df is already defined as per previous code\n\n# Step 1: Correlation Analysis to remove highly correlated features\ncorrelation_matrix = final_df.corr()\n\n# Set a threshold for correlation (e.g., 0.9) and remove highly correlated features\nthreshold = 0.9\nhighly_correlated = set()\nfor i in range(len(correlation_matrix.columns)):\n    for j in range(i):\n        if abs(correlation_matrix.iloc[i, j]) > threshold:\n            colname = correlation_matrix.columns[i]\n            highly_correlated.add(colname)\n\nprint(f\"Highly correlated features: {highly_correlated}\")\n\n# Drop highly correlated features from final_df\nfinal_df_filtered = final_df.drop(columns=highly_correlated)\n\n# Step 2: Variance Thresholding to remove features with low variance\n# Initialize VarianceThreshold with a threshold of 0.01 (default is 0)\nselector = VarianceThreshold(threshold=0.01)\nfinal_df_no_low_variance = selector.fit_transform(final_df_filtered)\n\n# Create a DataFrame with selected features\nfinal_df_no_low_variance = pd.DataFrame(final_df_no_low_variance, columns=final_df_filtered.columns[selector.get_support()])\n\n# Display the selected features\nprint(\"Selected features after variance thresholding:\")\nprint(final_df_no_low_variance.columns)\n\n# Optionally, you can inspect the selected features\nfinal_df_no_low_variance.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-15T03:49:53.852421Z","iopub.execute_input":"2025-04-15T03:49:53.853918Z","iopub.status.idle":"2025-04-15T03:49:54.14369Z","shell.execute_reply.started":"2025-04-15T03:49:53.85388Z","shell.execute_reply":"2025-04-15T03:49:54.142969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\nimport numpy as np\n\n# Assuming final_df_no_low_variance is your selected feature set for LSTM\nscaler = MinMaxScaler()\nscaled_data = scaler.fit_transform(final_df_no_low_variance)\n\n# Reshape data into 3D format: (samples, time steps, features)\nX = scaled_data\ny = scaled_data[:, 0]  # Assuming the first column is the target variable (e.g., 'pr_mean')\n\n# Reshaping the data for LSTM [samples, time steps, features]\nX = X.reshape((X.shape[0], 1, X.shape[1]))  # Adding time step dimension\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}