{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81000,"databundleVersionId":8812083,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-06T15:19:48.425859Z","iopub.execute_input":"2024-09-06T15:19:48.426358Z","iopub.status.idle":"2024-09-06T15:19:48.900809Z","shell.execute_reply.started":"2024-09-06T15:19:48.426313Z","shell.execute_reply":"2024-09-06T15:19:48.899514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Maize Files","metadata":{}},{"cell_type":"markdown","source":"## \"load_maize_train_files\" searches for files containing both \"maize\" and train in the filename, then loads them into a dictionary","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndef load_maize_train_files(folder_path):\n    # Dictionary to store the DataFrames\n    dataframes = {}\n    \n    # Iterate over files in the given folder\n    for filename in os.listdir(folder_path):\n        if \"maize\" in filename and \"train\" in filename and filename.endswith('.parquet'):\n            file_path = os.path.join(folder_path, filename)\n            \n            # Load the parquet file into a DataFrame\n            df_name = filename.replace('.parquet', '')  \n            dataframes[df_name] = pd.read_parquet(file_path)\n            print(f\"Loaded {filename} into {df_name}\")\n\n    return dataframes\n\nfolder_path = '/kaggle/input/the-future-crop-challenge/' \nmaize_data = load_maize_train_files(folder_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:23:50.725547Z","iopub.execute_input":"2024-09-06T15:23:50.726209Z","iopub.status.idle":"2024-09-06T15:24:22.131329Z","shell.execute_reply.started":"2024-09-06T15:23:50.726152Z","shell.execute_reply":"2024-09-06T15:24:22.130197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merging the datasets, join them by common columns \"ID\", \"LAT\", \"LON\", \"YEAR\", and \"CROP\"","metadata":{}},{"cell_type":"code","source":"# Merge train_solutions_maize with other files using 'ID'\nsolutions_df = maize_data['train_solutions_maize']\n\n# merge with other datasets\nkeys_to_merge = ['soil_co2_maize_train', 'rsds_maize_train', 'tasmin_maize_train', 'tasmax_maize_train', 'tas_maize_train', 'pr_maize_train']\n\nfor key in keys_to_merge:\n    merged_df = solutions_df.merge(maize_data[key], on='ID', how='left')\n    print(f\"Merged {key} into the main DataFrame\")\n\nprint(merged_df.info())\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:26:23.868482Z","iopub.execute_input":"2024-09-06T15:26:23.868953Z","iopub.status.idle":"2024-09-06T15:26:25.435719Z","shell.execute_reply.started":"2024-09-06T15:26:23.868906Z","shell.execute_reply":"2024-09-06T15:26:25.434532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"# Display the first few rows\nprint(merged_df.head())\n\n# Get a summary of the DataFramess\nprint(merged_df.info())\n\n# Summary statistics\nprint(merged_df.describe())\n\nprint(merged_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:26:51.287665Z","iopub.execute_input":"2024-09-06T15:26:51.288150Z","iopub.status.idle":"2024-09-06T15:26:56.251236Z","shell.execute_reply.started":"2024-09-06T15:26:51.288105Z","shell.execute_reply":"2024-09-06T15:26:56.249831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Checking for Missing Values","metadata":{}},{"cell_type":"code","source":"missing_values = merged_df.isnull().sum()\nprint(\"Missing Values:\\n\", missing_values)\n\n# visualize missing data\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.heatmap(merged_df.isnull(), cbar=False, cmap='viridis')\nplt.title('Missing Data Heatmap')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:29:14.481017Z","iopub.execute_input":"2024-09-06T15:29:14.482846Z","iopub.status.idle":"2024-09-06T15:30:40.531765Z","shell.execute_reply.started":"2024-09-06T15:29:14.482791Z","shell.execute_reply":"2024-09-06T15:30:40.530258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Calculating the mean of tas\nmerged_df['mean_tas'] = merged_df.loc[:, '0':'239'].mean(axis=1)\nmerged_df['mean_pr'] = merged_df.loc[:, '0':'239'].mean(axis=1)\nmerged_df['mean_rsds'] = merged_df.loc[:, '0':'239'].mean(axis=1)\n\n# Drop raw time series columns\nmerged_df = merged_df.drop(columns=[str(i) for i in range(240)])\n\n\nprint(merged_df.head())\nprint(merged_df.info())","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:31:05.084381Z","iopub.execute_input":"2024-09-06T15:31:05.085732Z","iopub.status.idle":"2024-09-06T15:31:06.875769Z","shell.execute_reply.started":"2024-09-06T15:31:05.085675Z","shell.execute_reply":"2024-09-06T15:31:06.874189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare & Split Data","metadata":{}},{"cell_type":"code","source":"X_maize = merged_df[['mean_tas', 'mean_pr', 'mean_rsds', 'lon', 'lat', 'year']]\ny_maize = merged_df['yield']\n\nfrom sklearn.model_selection import train_test_split\n\n# Split the data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X_maize, y_maize, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:31:59.517912Z","iopub.execute_input":"2024-09-06T15:31:59.518448Z","iopub.status.idle":"2024-09-06T15:31:59.577998Z","shell.execute_reply.started":"2024-09-06T15:31:59.518401Z","shell.execute_reply":"2024-09-06T15:31:59.576649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Model - Random Forest Model","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score\n\nrf_model = RandomForestRegressor(n_estimators=100, random_state=42)\n\n# Train model\nrf_model.fit(X_train, y_train)\n\ny_pred_rf = rf_model.predict(X_val)\n\n# Evaluate the model\nrmse_rf = mean_squared_error(y_val, y_pred_rf, squared=False)\nr2_rf = r2_score(y_val, y_pred_rf)\n\nprint(f\"Random Forest RMSE: {rmse_rf}\")\nprint(f\"Random Forest R-squared: {r2_rf}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:36:42.897879Z","iopub.execute_input":"2024-09-06T15:36:42.898360Z","iopub.status.idle":"2024-09-06T15:40:04.027845Z","shell.execute_reply.started":"2024-09-06T15:36:42.898315Z","shell.execute_reply":"2024-09-06T15:40:04.026507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Importance","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\n\n# Get feature importances from the Random Forest model\nimportances = rf_model.feature_importances_\n\n# DataFrame for better visualization\nfeature_names = X_train.columns\nimportance_df = pd.DataFrame({\n    'Feature': feature_names,\n    'Importance': importances\n})\n\n# Sort the DataFrame by importance\nimportance_df = importance_df.sort_values(by='Importance', ascending=False)\n\n# Plot the feature importances\nplt.figure(figsize=(10, 6))\nplt.barh(importance_df['Feature'], importance_df['Importance'])\nplt.xlabel('Importance')\nplt.ylabel('Feature')\nplt.title('Feature Importance - Random Forest')\nplt.gca().invert_yaxis()\nplt.show()\n\n# Display the importance DataFrame\nprint(importance_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:42:43.169413Z","iopub.execute_input":"2024-09-06T15:42:43.169843Z","iopub.status.idle":"2024-09-06T15:42:43.952995Z","shell.execute_reply.started":"2024-09-06T15:42:43.169806Z","shell.execute_reply":"2024-09-06T15:42:43.951478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The feature importance plot indicates that latitude (lat) and longitude (lon) are the most important features for predicting maize yield in your Random Forest model, with significantly higher importance compared to other features. This suggests that spatial location is a critical factor in determining yield, likely due to varying climate and soil conditions across different regions. Other features like year, mean_rsds (mean daily short-wave radiation), mean_pr (mean daily precipitation), and mean_tas (mean daily temperature) also contribute but to a lesser extent.\n\nSpatial location can significantly influence agricultural yield due to a variety of environmental and management factors that vary across different locations.","metadata":{}},{"cell_type":"markdown","source":"### Spatial Interactions\n\nSince lat and lon dominate, you may want to explore whether additional spatial features or interaction terms could further improve model accuracy.\n\nSpatial Interactions:\n\nCreate interaction terms between lat and lon (e.g., lat * lon).\nExplore polynomial features like lat^2, lon^2, or their combinations to capture more complex spatial relationships.","metadata":{}},{"cell_type":"code","source":"# Create interaction terms between lat and lon\nmerged_df['lat_lon_interaction'] = merged_df['lat'] * merged_df['lon']\nmerged_df['lat_squared'] = merged_df['lat'] ** 2\nmerged_df['lon_squared'] = merged_df['lon'] ** 2\n\n# Update the feature set\nX_maize = merged_df[['mean_tas', 'mean_pr', 'mean_rsds', 'lon', 'lat', 'year', 'lat_lon_interaction', 'lat_squared', 'lon_squared']]\n\n# Split the data\nX_train, X_val, y_train, y_val = train_test_split(X_maize, y_maize, test_size=0.2, random_state=42)\n\n# Train the Random Forest model with the new features\nrf_model.fit(X_train, y_train)\ny_pred_rf = rf_model.predict(X_val)\n\n# Evaluate the model\nrmse_rf = mean_squared_error(y_val, y_pred_rf, squared=False)\nr2_rf = r2_score(y_val, y_pred_rf)\n\nprint(f\"Updated Random Forest RMSE: {rmse_rf}\")\nprint(f\"Updated Random Forest R-squared: {r2_rf}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T15:42:50.163627Z","iopub.execute_input":"2024-09-06T15:42:50.164490Z","iopub.status.idle":"2024-09-06T15:47:06.186525Z","shell.execute_reply.started":"2024-09-06T15:42:50.164440Z","shell.execute_reply":"2024-09-06T15:47:06.185282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explanation:\nlat_lon_interaction: Captures the interaction between latitude and longitude.\n\nlat_squared and lon_squared: Allows the model to capture non-linear spatial effects.","metadata":{}}]}