{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81000,"databundleVersionId":8812083,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns # plotting\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-13T04:45:23.212305Z","iopub.execute_input":"2024-06-13T04:45:23.212696Z","iopub.status.idle":"2024-06-13T04:45:24.241579Z","shell.execute_reply.started":"2024-06-13T04:45:23.212663Z","shell.execute_reply":"2024-06-13T04:45:24.240669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import RandomForestRegressor\n\n# Function to load data\ndef load_data(crop: str, mode: str=\"train\"):\n    base_path = \"/kaggle/input/the-future-crop-challenge\"\n\n    tasmax = pd.read_parquet(f\"{base_path}/tasmax_{crop}_{mode}.parquet\")\n    tasmin = pd.read_parquet(f\"{base_path}/tasmin_{crop}_{mode}.parquet\")\n    pr = pd.read_parquet(f\"{base_path}/pr_{crop}_{mode}.parquet\")\n    rsds = pd.read_parquet(f\"{base_path}/rsds_{crop}_{mode}.parquet\")\n    soil_co2 = pd.read_parquet(f\"{base_path}/soil_co2_{crop}_{mode}.parquet\")\n\n    target = None\n    if mode == \"train\":\n        target = pd.read_parquet(f\"{base_path}/train_solutions_{crop}.parquet\")\n\n    return {\n        'tasmax': tasmax,\n        'tasmin': tasmin,\n        'pr': pr,\n        'rsds': rsds,\n        'soil_co2': soil_co2,\n        'target': target,\n    }\n\n# Load wheat data\nwheat_train = load_data(\"wheat\", \"train\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T04:45:24.360086Z","iopub.execute_input":"2024-06-13T04:45:24.360364Z","iopub.status.idle":"2024-06-13T04:45:48.021555Z","shell.execute_reply.started":"2024-06-13T04:45:24.360341Z","shell.execute_reply":"2024-06-13T04:45:48.020515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Aggregate daily data\ndef aggregate_daily_data(df):\n    numeric_df = df.iloc[:, 4:].apply(pd.to_numeric, errors='coerce')\n    return numeric_df.agg(['mean', 'max', 'min', 'std'], axis=1)\n\ntasmax_agg_wheat = aggregate_daily_data(wheat_train['tasmax'])\ntasmin_agg_wheat = aggregate_daily_data(wheat_train['tasmin'])\npr_agg_wheat = aggregate_daily_data(wheat_train['pr'])\nrsds_agg_wheat = aggregate_daily_data(wheat_train['rsds'])\n\n# Merge datasets\nfeatures_wheat = pd.concat([tasmax_agg_wheat, tasmin_agg_wheat, pr_agg_wheat, rsds_agg_wheat, wheat_train['soil_co2']], axis=1)\ntarget_wheat = wheat_train['target']\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T04:45:48.024596Z","iopub.execute_input":"2024-06-13T04:45:48.025386Z","iopub.status.idle":"2024-06-13T05:04:40.591065Z","shell.execute_reply.started":"2024-06-13T04:45:48.025357Z","shell.execute_reply":"2024-06-13T05:04:40.589878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop non-numeric columns\nnumeric_features_wheat = features_wheat.select_dtypes(include=[np.number])\n\n# Scale features\nscaler_wheat = StandardScaler()\nfeatures_scaled_wheat = scaler_wheat.fit_transform(numeric_features_wheat)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:04:40.592393Z","iopub.execute_input":"2024-06-13T05:04:40.592722Z","iopub.status.idle":"2024-06-13T05:04:40.769333Z","shell.execute_reply.started":"2024-06-13T05:04:40.592679Z","shell.execute_reply":"2024-06-13T05:04:40.768298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(features_scaled_wheat, target_wheat, test_size=0.2, random_state=42)\ny_train = y_train.values.ravel()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:04:40.770509Z","iopub.execute_input":"2024-06-13T05:04:40.770829Z","iopub.status.idle":"2024-06-13T05:04:40.852575Z","shell.execute_reply.started":"2024-06-13T05:04:40.770804Z","shell.execute_reply":"2024-06-13T05:04:40.851532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RandomForestRegressor(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:04:40.854515Z","iopub.execute_input":"2024-06-13T05:04:40.854811Z","iopub.status.idle":"2024-06-13T05:13:44.954371Z","shell.execute_reply.started":"2024-06-13T05:04:40.854787Z","shell.execute_reply":"2024-06-13T05:13:44.953402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\n# Evaluation\ny_pred = model.predict(X_val)\nrmse = mean_squared_error(y_val, y_pred, squared=False)\nprint(f\"RMSE: {rmse}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:13:44.955498Z","iopub.execute_input":"2024-06-13T05:13:44.955791Z","iopub.status.idle":"2024-06-13T05:13:47.729510Z","shell.execute_reply.started":"2024-06-13T05:13:44.955767Z","shell.execute_reply":"2024-06-13T05:13:47.728495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load test data\nwheat_test = load_data(\"wheat\", \"test\")\n\n# Aggregate daily data for test set\ntasmax_agg_test = aggregate_daily_data(wheat_test['tasmax'])\ntasmin_agg_test = aggregate_daily_data(wheat_test['tasmin'])\npr_agg_test = aggregate_daily_data(wheat_test['pr'])\nrsds_agg_test = aggregate_daily_data(wheat_test['rsds'])\n\n# Merge test datasets\nfeatures_test = pd.concat([tasmax_agg_test, tasmin_agg_test, pr_agg_test, rsds_agg_test, wheat_test['soil_co2']], axis=1)\n\n# Drop non-numeric columns from test set\nnumeric_features_test = features_test.select_dtypes(include=[np.number])\n\n# Drop rows with NaN values in the test set\n# numeric_features_test = numeric_features_test.dropna()\n\n# Scale features\nfeatures_scaled_test = scaler_wheat.transform(numeric_features_test)\n\n# Ensure to match the index with the corresponding IDs\ntest_ids = numeric_features_test.index\n\n# Make predictions\npredictions = model.predict(features_scaled_test)\n\n# Create a DataFrame with the IDs and predictions\nsubmission_df = pd.DataFrame({\n    'ID': test_ids,\n    'yield': predictions\n})\n\n# Save the DataFrame to a CSV file\nsubmission_df.to_csv('submission_wheat.csv', index=False)\n\n# first few rows of the submission file to verify\nprint(submission_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:13:47.730596Z","iopub.execute_input":"2024-06-13T05:13:47.730895Z","iopub.status.idle":"2024-06-13T05:52:06.556660Z","shell.execute_reply.started":"2024-06-13T05:13:47.730871Z","shell.execute_reply":"2024-06-13T05:52:06.555590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}