{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81000,"databundleVersionId":8812083,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:24:59.452052Z","iopub.execute_input":"2025-05-28T05:24:59.452305Z","iopub.status.idle":"2025-05-28T05:24:59.471585Z","shell.execute_reply.started":"2025-05-28T05:24:59.452288Z","shell.execute_reply":"2025-05-28T05:24:59.470899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rsds = pd.read_parquet(\"/kaggle/input/the-future-crop-challenge/rsds_wheat_train.parquet\")\npr = pd.read_parquet(\"/kaggle/input/the-future-crop-challenge/pr_wheat_train.parquet\")\ntas = pd.read_parquet(\"/kaggle/input/the-future-crop-challenge/tas_wheat_train.parquet\")\ntmin = pd.read_parquet(\"/kaggle/input/the-future-crop-challenge/tasmin_wheat_train.parquet\")\ntmax = pd.read_parquet(\"/kaggle/input/the-future-crop-challenge/tasmax_wheat_train.parquet\")\nsoilco2 = pd.read_parquet(\"/kaggle/input/the-future-crop-challenge/soil_co2_wheat_train.parquet\")\nyeild = pd.read_parquet(\"/kaggle/input/the-future-crop-challenge/train_solutions_wheat.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:24:59.472329Z","iopub.execute_input":"2025-05-28T05:24:59.472528Z","iopub.status.idle":"2025-05-28T05:25:18.421214Z","shell.execute_reply.started":"2025-05-28T05:24:59.472510Z","shell.execute_reply":"2025-05-28T05:25:18.420433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rsds.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.422049Z","iopub.execute_input":"2025-05-28T05:25:18.422770Z","iopub.status.idle":"2025-05-28T05:25:18.457930Z","shell.execute_reply.started":"2025-05-28T05:25:18.422744Z","shell.execute_reply":"2025-05-28T05:25:18.457377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rsds.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.459656Z","iopub.execute_input":"2025-05-28T05:25:18.459826Z","iopub.status.idle":"2025-05-28T05:25:18.480944Z","shell.execute_reply.started":"2025-05-28T05:25:18.459813Z","shell.execute_reply":"2025-05-28T05:25:18.480421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pr.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.481568Z","iopub.execute_input":"2025-05-28T05:25:18.481822Z","iopub.status.idle":"2025-05-28T05:25:18.499790Z","shell.execute_reply.started":"2025-05-28T05:25:18.481800Z","shell.execute_reply":"2025-05-28T05:25:18.499167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tas.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.500493Z","iopub.execute_input":"2025-05-28T05:25:18.500734Z","iopub.status.idle":"2025-05-28T05:25:18.522840Z","shell.execute_reply.started":"2025-05-28T05:25:18.500717Z","shell.execute_reply":"2025-05-28T05:25:18.522229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmin.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.523827Z","iopub.execute_input":"2025-05-28T05:25:18.524053Z","iopub.status.idle":"2025-05-28T05:25:18.545632Z","shell.execute_reply.started":"2025-05-28T05:25:18.524029Z","shell.execute_reply":"2025-05-28T05:25:18.544942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmax.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.546299Z","iopub.execute_input":"2025-05-28T05:25:18.546487Z","iopub.status.idle":"2025-05-28T05:25:18.567669Z","shell.execute_reply.started":"2025-05-28T05:25:18.546474Z","shell.execute_reply":"2025-05-28T05:25:18.566917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"soilco2.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.568443Z","iopub.execute_input":"2025-05-28T05:25:18.568735Z","iopub.status.idle":"2025-05-28T05:25:18.584770Z","shell.execute_reply.started":"2025-05-28T05:25:18.568711Z","shell.execute_reply":"2025-05-28T05:25:18.584006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"soilco2.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.585486Z","iopub.execute_input":"2025-05-28T05:25:18.585743Z","iopub.status.idle":"2025-05-28T05:25:18.625155Z","shell.execute_reply.started":"2025-05-28T05:25:18.585726Z","shell.execute_reply":"2025-05-28T05:25:18.624600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"soilco2.reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.625769Z","iopub.execute_input":"2025-05-28T05:25:18.625980Z","iopub.status.idle":"2025-05-28T05:25:18.643184Z","shell.execute_reply.started":"2025-05-28T05:25:18.625942Z","shell.execute_reply":"2025-05-28T05:25:18.642615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yeild.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.643779Z","iopub.execute_input":"2025-05-28T05:25:18.644032Z","iopub.status.idle":"2025-05-28T05:25:18.650366Z","shell.execute_reply.started":"2025-05-28T05:25:18.644016Z","shell.execute_reply":"2025-05-28T05:25:18.649686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def summarize_time_series(df, id_cols, prefix):\n    # If ID is in index, reset it to bring as column(s)\n    if not all(col in df.columns for col in id_cols):\n        df = df.reset_index()\n\n    # Select numeric columns excluding id_cols\n    ts_cols = df.drop(columns=id_cols, errors='ignore').select_dtypes(include='number').columns.tolist()\n\n    # Copy ID columns for summary_df\n    summary_df = df[id_cols].copy()\n\n    # Compute summary statistics row-wise for ts_cols\n    summary_df[f'{prefix}_mean'] = df[ts_cols].mean(axis=1)\n    summary_df[f'{prefix}_std'] = df[ts_cols].std(axis=1)\n    summary_df[f'{prefix}_min'] = df[ts_cols].min(axis=1)\n    summary_df[f'{prefix}_max'] = df[ts_cols].max(axis=1)\n    summary_df[f'{prefix}_median'] = df[ts_cols].median(axis=1)\n    summary_df[f'{prefix}_trend'] = df[ts_cols].apply(lambda row: row.diff().mean(), axis=1)\n\n    return summary_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.651105Z","iopub.execute_input":"2025-05-28T05:25:18.651347Z","iopub.status.idle":"2025-05-28T05:25:18.660451Z","shell.execute_reply.started":"2025-05-28T05:25:18.651325Z","shell.execute_reply":"2025-05-28T05:25:18.659938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define your ID columns\nid_cols = ['ID','crop','year', 'lat', 'lon']\n\n# File mapping (update file paths accordingly)\nfiles = {\n    'precip': '/kaggle/input/the-future-crop-challenge/pr_wheat_train.parquet',\n    'srad': '/kaggle/input/the-future-crop-challenge/rsds_wheat_train.parquet',\n    'tmean': '/kaggle/input/the-future-crop-challenge/tas_wheat_train.parquet',\n    'tmax': '/kaggle/input/the-future-crop-challenge/tasmax_wheat_train.parquet',\n    'tmin': '/kaggle/input/the-future-crop-challenge/tasmin_wheat_train.parquet'\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.663409Z","iopub.execute_input":"2025-05-28T05:25:18.663590Z","iopub.status.idle":"2025-05-28T05:25:18.671728Z","shell.execute_reply.started":"2025-05-28T05:25:18.663577Z","shell.execute_reply":"2025-05-28T05:25:18.671203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"summarized_dfs = []\n\nfor prefix, file in files.items():\n    print(f\"Processing: {file}\")\n    df = pd.read_parquet(file)\n\n    # Reset index if needed (your function also does this)\n    if df.index.name == 'ID' and 'ID' not in df.columns:\n        df = df.reset_index()\n\n    summary = summarize_time_series(df, id_cols=id_cols, prefix=prefix)\n    summarized_dfs.append(summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:25:18.672492Z","iopub.execute_input":"2025-05-28T05:25:18.673174Z","iopub.status.idle":"2025-05-28T05:27:41.072590Z","shell.execute_reply.started":"2025-05-28T05:25:18.673153Z","shell.execute_reply":"2025-05-28T05:27:41.071657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from functools import reduce\nfinal_summary = reduce(lambda left, right: pd.merge(left, right, on=id_cols, how='outer'), summarized_dfs)\nfinal_summary.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:41.073480Z","iopub.execute_input":"2025-05-28T05:27:41.073753Z","iopub.status.idle":"2025-05-28T05:27:41.782472Z","shell.execute_reply.started":"2025-05-28T05:27:41.073726Z","shell.execute_reply":"2025-05-28T05:27:41.781734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming final_summary is the combined summary DataFrame from before\nsoil = pd.read_parquet(\"/kaggle/input/the-future-crop-challenge/soil_co2_wheat_train.parquet\")\n\n# If soil DataFrame has ID as index, reset index\nif soil.index.name == 'ID' and 'ID' not in soil.columns:\n    soil = soil.reset_index()\n\n# Check if all id_cols are in soil\nmissing_cols = [col for col in id_cols if col not in soil.columns]\nif missing_cols:\n    print(f\"Warning: Missing columns in soil dataset: {missing_cols}\")\n\n# Merge soil data with final summary on id_cols\nfinal_df = pd.merge(final_summary, soil, on=id_cols, how='inner')\n\nfinal_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:41.783295Z","iopub.execute_input":"2025-05-28T05:27:41.783558Z","iopub.status.idle":"2025-05-28T05:27:41.981849Z","shell.execute_reply.started":"2025-05-28T05:27:41.783534Z","shell.execute_reply":"2025-05-28T05:27:41.981111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:41.982709Z","iopub.execute_input":"2025-05-28T05:27:41.982902Z","iopub.status.idle":"2025-05-28T05:27:42.034408Z","shell.execute_reply.started":"2025-05-28T05:27:41.982881Z","shell.execute_reply":"2025-05-28T05:27:42.033626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yield_df = pd.read_parquet('/kaggle/input/the-future-crop-challenge/train_solutions_wheat.parquet')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:42.035095Z","iopub.execute_input":"2025-05-28T05:27:42.035383Z","iopub.status.idle":"2025-05-28T05:27:42.050341Z","shell.execute_reply.started":"2025-05-28T05:27:42.035360Z","shell.execute_reply":"2025-05-28T05:27:42.049703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# If yield_df index is ID and not a column, reset it\nif yield_df.index.name == 'ID' and 'ID' not in yield_df.columns:\n    yield_df = yield_df.reset_index()\n\n# Merge on common keys\nm_df = pd.merge(final_df, yield_df, on=['ID'], how='inner')\n\nm_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:42.051050Z","iopub.execute_input":"2025-05-28T05:27:42.051268Z","iopub.status.idle":"2025-05-28T05:27:42.157572Z","shell.execute_reply.started":"2025-05-28T05:27:42.051240Z","shell.execute_reply":"2025-05-28T05:27:42.156770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"m_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:42.158473Z","iopub.execute_input":"2025-05-28T05:27:42.158762Z","iopub.status.idle":"2025-05-28T05:27:42.211347Z","shell.execute_reply.started":"2025-05-28T05:27:42.158738Z","shell.execute_reply":"2025-05-28T05:27:42.210760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"m_df = m_df.drop(columns='crop')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:42.212095Z","iopub.execute_input":"2025-05-28T05:27:42.212376Z","iopub.status.idle":"2025-05-28T05:27:42.241343Z","shell.execute_reply.started":"2025-05-28T05:27:42.212352Z","shell.execute_reply":"2025-05-28T05:27:42.240593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"m_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:42.242121Z","iopub.execute_input":"2025-05-28T05:27:42.242811Z","iopub.status.idle":"2025-05-28T05:27:42.261613Z","shell.execute_reply.started":"2025-05-28T05:27:42.242787Z","shell.execute_reply":"2025-05-28T05:27:42.260912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"m_df.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:42.262261Z","iopub.execute_input":"2025-05-28T05:27:42.262530Z","iopub.status.idle":"2025-05-28T05:27:42.274434Z","shell.execute_reply.started":"2025-05-28T05:27:42.262503Z","shell.execute_reply":"2025-05-28T05:27:42.273867Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Distribution of Yield\nWhy: Understand overall productivity, detect outliers","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.histplot(m_df['yield'], kde=True)\nplt.title(\"Distribution of Yield\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:42.275084Z","iopub.execute_input":"2025-05-28T05:27:42.275272Z","iopub.status.idle":"2025-05-28T05:27:44.266061Z","shell.execute_reply.started":"2025-05-28T05:27:42.275259Z","shell.execute_reply":"2025-05-28T05:27:44.265278Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Correlation with Yield\nWhy: Know which variables most affect yield","metadata":{}},{"cell_type":"code","source":"sns.heatmap(m_df.corr(numeric_only=True), fmt=\".2f\", cmap='coolwarm')\nplt.title(\"Feature Correlation Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:44.266685Z","iopub.execute_input":"2025-05-28T05:27:44.266993Z","iopub.status.idle":"2025-05-28T05:27:45.678953Z","shell.execute_reply.started":"2025-05-28T05:27:44.266978Z","shell.execute_reply":"2025-05-28T05:27:45.678295Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Geospatial Analysis (Yield by Location)\nWhy: Know if some locations are more productive","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsc = plt.scatter(m_df['lon'], m_df['lat'], c=m_df['yield'], cmap='YlGn', alpha=0.7)\nplt.colorbar(sc, label='Yield')\nplt.xlabel(\"Longitude\")\nplt.ylabel(\"Latitude\")\nplt.title(\"Geospatial Distribution of Yield\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:45.679800Z","iopub.execute_input":"2025-05-28T05:27:45.680121Z","iopub.status.idle":"2025-05-28T05:27:50.194480Z","shell.execute_reply.started":"2025-05-28T05:27:45.680089Z","shell.execute_reply":"2025-05-28T05:27:50.193784Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  4. Temporal Trends (Year-wise Yield)\nWhy: Detect climate effects or improvements over time","metadata":{}},{"cell_type":"code","source":"sns.lineplot(data=m_df, x='year', y='yield')\nplt.title(\"Year-wise Yield Trend\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:50.195268Z","iopub.execute_input":"2025-05-28T05:27:50.195843Z","iopub.status.idle":"2025-05-28T05:27:52.890097Z","shell.execute_reply.started":"2025-05-28T05:27:50.195817Z","shell.execute_reply":"2025-05-28T05:27:52.889284Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Yield vs CO₂ / Nitrogen\nWhy: Direct relation with inputs and climate","metadata":{}},{"cell_type":"code","source":"sns.scatterplot(data=m_df, x='co2', y='yield')\nplt.title(\"CO2 vs Yield\")\n\nsns.scatterplot(data=m_df, x='nitrogen', y='yield')\nplt.title(\"Nitrogen vs Yield\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:52.890846Z","iopub.execute_input":"2025-05-28T05:27:52.891153Z","iopub.status.idle":"2025-05-28T05:27:54.011504Z","shell.execute_reply.started":"2025-05-28T05:27:52.891135Z","shell.execute_reply":"2025-05-28T05:27:54.010714Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Texture Class Impact on Yield\nWhy: Soil quality may affect growth","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=m_df, x='texture_class', y='yield')\nplt.title(\"Soil Texture Class vs Yield\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:54.012224Z","iopub.execute_input":"2025-05-28T05:27:54.012428Z","iopub.status.idle":"2025-05-28T05:27:54.348836Z","shell.execute_reply.started":"2025-05-28T05:27:54.012412Z","shell.execute_reply":"2025-05-28T05:27:54.348119Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  7. Compare Top Features from Summary Stats\nYou summarized time series features into mean, std, etc. Check their impact:","metadata":{}},{"cell_type":"code","source":"summary_cols = [col for col in m_df.columns if any(stat in col for stat in ['mean', 'std', 'trend'])]\nfor col in summary_cols:\n    sns.scatterplot(data=m_df, x=col, y='yield')\n    plt.title(f\"{col} vs Yield\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:27:54.349590Z","iopub.execute_input":"2025-05-28T05:27:54.349784Z","iopub.status.idle":"2025-05-28T05:28:05.903026Z","shell.execute_reply.started":"2025-05-28T05:27:54.349769Z","shell.execute_reply":"2025-05-28T05:28:05.902283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"m_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:28:05.903768Z","iopub.execute_input":"2025-05-28T05:28:05.904017Z","iopub.status.idle":"2025-05-28T05:28:05.923440Z","shell.execute_reply.started":"2025-05-28T05:28:05.903994Z","shell.execute_reply":"2025-05-28T05:28:05.922763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Drop non-feature columns\ndrop_cols = ['id', 'lat', 'lon','yield']  # 'yield' is target\nfeature_cols = [col for col in m_df.columns if col not in drop_cols]\n\n# Define X (features) and y (target)\nX = m_df[feature_cols]\ny = m_df['yield']\n\n# Train-test split (80-20)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nprint(\"Train shape:\", X_train.shape)\nprint(\"Test shape:\", X_test.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:28:05.924182Z","iopub.execute_input":"2025-05-28T05:28:05.924508Z","iopub.status.idle":"2025-05-28T05:28:06.038429Z","shell.execute_reply.started":"2025-05-28T05:28:05.924485Z","shell.execute_reply":"2025-05-28T05:28:06.037792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:28:06.039115Z","iopub.execute_input":"2025-05-28T05:28:06.039318Z","iopub.status.idle":"2025-05-28T05:28:06.056510Z","shell.execute_reply.started":"2025-05-28T05:28:06.039301Z","shell.execute_reply":"2025-05-28T05:28:06.055862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:28:06.057168Z","iopub.execute_input":"2025-05-28T05:28:06.057362Z","iopub.status.idle":"2025-05-28T05:28:06.068225Z","shell.execute_reply.started":"2025-05-28T05:28:06.057346Z","shell.execute_reply":"2025-05-28T05:28:06.067446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\nimport numpy as np\n\n# Initialize and train model\nrf_model = RandomForestRegressor(n_estimators=100, random_state=42)\nrf_model.fit(X_train, y_train)\n\n# Predict\ny_pred = rf_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T06:42:48.302764Z","iopub.execute_input":"2025-05-26T06:42:48.303134Z","iopub.status.idle":"2025-05-26T07:02:58.233387Z","shell.execute_reply.started":"2025-05-26T06:42:48.303108Z","shell.execute_reply":"2025-05-26T07:02:58.232360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluation metrics\nrmse = np.sqrt(mean_squared_error(y_test, y_pred))\nmae = mean_absolute_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"📊 RMSE: {rmse:.4f}\")\nprint(f\"📉 MAE: {mae:.4f}\")\nprint(f\"📈 R² Score: {r2:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T07:02:58.274404Z","iopub.execute_input":"2025-05-26T07:02:58.274747Z","iopub.status.idle":"2025-05-26T07:02:58.291298Z","shell.execute_reply.started":"2025-05-26T07:02:58.274716Z","shell.execute_reply":"2025-05-26T07:02:58.290433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nimportances = rf_model.feature_importances_\nfeatures = X_train.columns\n\n# Sort features by importance\nindices = np.argsort(importances)[::-1]\n\n# Plot\nplt.figure(figsize=(10, 6))\nplt.title(\"Feature Importances\")\nplt.bar(range(len(features)), importances[indices], align='center')\nplt.xticks(range(len(features)), [features[i] for i in indices], rotation=90)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T07:02:58.292238Z","iopub.execute_input":"2025-05-26T07:02:58.292460Z","iopub.status.idle":"2025-05-26T07:02:59.071755Z","shell.execute_reply.started":"2025-05-26T07:02:58.292442Z","shell.execute_reply":"2025-05-26T07:02:59.070830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression, Ridge, Lasso\nfrom sklearn.svm import SVR\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\nimport numpy as np\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:28:06.068938Z","iopub.execute_input":"2025-05-28T05:28:06.069224Z","iopub.status.idle":"2025-05-28T05:28:06.082136Z","shell.execute_reply.started":"2025-05-28T05:28:06.069210Z","shell.execute_reply":"2025-05-28T05:28:06.081478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = {\n    \"Linear Regression\": LinearRegression(),\n    \"Ridge\": Ridge(alpha=1.0),\n    \"Lasso\": Lasso(alpha=0.1),\n    \"Decision Tree\": DecisionTreeRegressor(random_state=42),\n    \"Random Forest\": RandomForestRegressor(n_estimators=100, random_state=42),\n    \"XGBoost\": XGBRegressor(n_estimators=100, learning_rate=0.1, max_depth=6, random_state=42)\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:28:06.082702Z","iopub.execute_input":"2025-05-28T05:28:06.082936Z","iopub.status.idle":"2025-05-28T05:28:06.097029Z","shell.execute_reply.started":"2025-05-28T05:28:06.082916Z","shell.execute_reply":"2025-05-28T05:28:06.096434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = []\n\nfor name, model in models.items():\n    model.fit(X_train, y_train)\n    y_pred = model.predict(X_test)\n\n    rmse = np.sqrt(mean_squared_error(y_test, y_pred))\n    mae = mean_absolute_error(y_test, y_pred)\n    r2 = r2_score(y_test, y_pred)\n\n    results.append((name, rmse, mae, r2))\n\nresults_df = pd.DataFrame(results, columns=[\"Model\", \"RMSE\", \"MAE\", \"R2_Score\"])\nresults_df.sort_values(by=\"R2_Score\", ascending=False, inplace=True)\nresults_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T08:04:59.390415Z","iopub.execute_input":"2025-05-26T08:04:59.391120Z","iopub.status.idle":"2025-05-26T08:21:34.927609Z","shell.execute_reply.started":"2025-05-26T08:04:59.391096Z","shell.execute_reply":"2025-05-26T08:21:34.926832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.barplot(data=results_df, x=\"R2_Score\", y=\"Model\", palette=\"viridis\")\nplt.title(\"Model Comparison Based on R² Score\", fontsize=14)\nplt.xlabel(\"R² Score\")\nplt.ylabel(\"Model\")\nplt.grid(True, axis='x')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T08:21:34.928442Z","iopub.execute_input":"2025-05-26T08:21:34.928685Z","iopub.status.idle":"2025-05-26T08:21:35.142336Z","shell.execute_reply.started":"2025-05-26T08:21:34.928666Z","shell.execute_reply":"2025-05-26T08:21:35.141695Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Hyperparameter Tuning\n\n#### RandomForest","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\nfrom sklearn.ensemble import RandomForestRegressor\n\nparam_dist = {\n    'n_estimators': [100, 200, 300],\n    'max_depth': [10, 20, 30, None],\n    'min_samples_split': [2, 5, 10],\n    'min_samples_leaf': [1, 2, 4],\n    'max_features': ['sqrt', 'log2']\n}\n\nrf = RandomForestRegressor(random_state=42, n_jobs=-1)\n\nsearch = RandomizedSearchCV(rf, param_distributions=param_dist, \n                            n_iter=20, cv=3, scoring='neg_root_mean_squared_error', verbose=2)\nsearch.fit(X_train, y_train)\n\nprint(\"Best RF params:\", search.best_params_)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T05:28:06.100059Z","iopub.execute_input":"2025-05-28T05:28:06.100257Z","iopub.status.idle":"2025-05-28T06:37:27.224323Z","shell.execute_reply.started":"2025-05-28T05:28:06.100240Z","shell.execute_reply":"2025-05-28T06:37:27.223488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nimportances = search.best_estimator_.feature_importances_\nfeat_names = X_train.columns\n\nplt.figure(figsize=(10, 6))\nplt.barh(feat_names, importances)\nplt.xlabel(\"Importance\")\nplt.title(\"Feature Importance (Random Forest)\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T06:37:27.225254Z","iopub.execute_input":"2025-05-28T06:37:27.225523Z","iopub.status.idle":"2025-05-28T06:37:27.831586Z","shell.execute_reply.started":"2025-05-28T06:37:27.225496Z","shell.execute_reply":"2025-05-28T06:37:27.830904Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Xgboost","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom sklearn.model_selection import RandomizedSearchCV\n\n# Define the model\nxgb = XGBRegressor(objective='reg:squarederror', random_state=42, n_jobs=-1)\n\n# Parameter grid\nparam_dist = {\n    'n_estimators': [100, 200, 300],\n    'max_depth': [3, 5, 7, 10],\n    'learning_rate': [0.01, 0.05, 0.1, 0.3],\n    'subsample': [0.6, 0.8, 1.0],\n    'colsample_bytree': [0.6, 0.8, 1.0],\n    'gamma': [0, 0.1, 0.2, 0.3],\n    'reg_alpha': [0, 0.1, 0.5],\n    'reg_lambda': [1, 1.5, 2.0]\n}\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T06:37:39.114530Z","iopub.execute_input":"2025-05-28T06:37:39.115148Z","iopub.status.idle":"2025-05-28T06:37:39.121932Z","shell.execute_reply.started":"2025-05-28T06:37:39.115116Z","shell.execute_reply":"2025-05-28T06:37:39.120818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Randomized Search\nxgb_search = RandomizedSearchCV(\n    estimator=xgb,\n    param_distributions=param_dist,\n    n_iter=25,\n    scoring='neg_root_mean_squared_error',\n    cv=3,\n    verbose=2,\n    random_state=42\n)\n\n# Fit the search\nxgb_search.fit(X_train, y_train)\n\n# Best params and score\nprint(\"✅ Best XGBoost Parameters:\")\nprint(xgb_search.best_params_)\n\nprint(\"📉 Best RMSE Score:\")\nprint(-xgb_search.best_score_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T06:37:52.602574Z","iopub.execute_input":"2025-05-28T06:37:52.603251Z","iopub.status.idle":"2025-05-28T06:43:32.165370Z","shell.execute_reply.started":"2025-05-28T06:37:52.603226Z","shell.execute_reply":"2025-05-28T06:43:32.164582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\n\nbest_rf_model = search.best_estimator_\nbest_xgb_model = xgb_search.best_estimator_\n\n# Predict with tuned models\nrf_preds = best_rf_model.predict(X_test)\nxgb_preds = best_xgb_model.predict(X_test)\n\n# Evaluate\ndef evaluate_model(y_true, y_pred, name=\"Model\"):\n    print(f\"📊 {name} Evaluation:\")\n    print(f\"RMSE: {mean_squared_error(y_true, y_pred, squared=False):.4f}\")\n    print(f\"MAE: {mean_absolute_error(y_true, y_pred):.4f}\")\n    print(f\"R² Score: {r2_score(y_true, y_pred):.4f}\")\n\nevaluate_model(y_test, rf_preds, \"Random Forest\")\nevaluate_model(y_test, xgb_preds, \"XGBoost\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T06:50:56.303452Z","iopub.execute_input":"2025-05-28T06:50:56.303910Z","iopub.status.idle":"2025-05-28T06:50:59.277257Z","shell.execute_reply.started":"2025-05-28T06:50:56.303888Z","shell.execute_reply":"2025-05-28T06:50:59.276113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\njoblib.dump(best_rf_model, \"best_random_forest_model.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T06:54:11.574336Z","iopub.execute_input":"2025-05-28T06:54:11.574608Z","iopub.status.idle":"2025-05-28T06:54:18.640508Z","shell.execute_reply.started":"2025-05-28T06:54:11.574589Z","shell.execute_reply":"2025-05-28T06:54:18.639739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.to_csv('/kaggle/working/X_train.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:19:06.530091Z","iopub.execute_input":"2025-05-28T07:19:06.530378Z","iopub.status.idle":"2025-05-28T07:19:15.125116Z","shell.execute_reply.started":"2025-05-28T07:19:06.530356Z","shell.execute_reply":"2025-05-28T07:19:15.124530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns\n\n# Assuming your training features are in X_train\nimportances = best_rf_model.feature_importances_\nfeatures = X_train.columns\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importances}).sort_values(by='Importance', ascending=False)\n\nplt.figure(figsize=(10, 6))\nsns.barplot(x='Importance', y='Feature', data=importance_df.head(15))\nplt.title(\"Top 15 Important Features (Random Forest)\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T06:55:16.864927Z","iopub.execute_input":"2025-05-28T06:55:16.865235Z","iopub.status.idle":"2025-05-28T06:55:17.404417Z","shell.execute_reply.started":"2025-05-28T06:55:16.865215Z","shell.execute_reply":"2025-05-28T06:55:17.403720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import r2_score\nimport matplotlib.pyplot as plt\n\ny_pred = best_rf_model.predict(X_test)\n\nplt.figure(figsize=(6, 6))\nplt.scatter(y_test, y_pred, alpha=0.3)\nplt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'r--')\nplt.xlabel('Actual Yield')\nplt.ylabel('Predicted Yield')\nplt.title(f'Actual vs Predicted Yield (R² = {r2_score(y_test, y_pred):.3f})')\nplt.grid(True)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T06:56:09.275749Z","iopub.execute_input":"2025-05-28T06:56:09.276080Z","iopub.status.idle":"2025-05-28T06:56:12.125714Z","shell.execute_reply.started":"2025-05-28T06:56:09.276057Z","shell.execute_reply":"2025-05-28T06:56:12.124940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T06:59:29.611671Z","iopub.execute_input":"2025-05-28T06:59:29.612381Z","iopub.status.idle":"2025-05-28T06:59:29.616653Z","shell.execute_reply.started":"2025-05-28T06:59:29.612351Z","shell.execute_reply":"2025-05-28T06:59:29.615849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:02:02.666474Z","iopub.execute_input":"2025-05-28T07:02:02.667075Z","iopub.status.idle":"2025-05-28T07:02:02.671874Z","shell.execute_reply.started":"2025-05-28T07:02:02.667052Z","shell.execute_reply":"2025-05-28T07:02:02.671042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Get feature importances from the best model\nimportances = best_rf_model.feature_importances_\nfeature_names = X_train.columns  # Assuming you used pandas DataFrame\n\n# Create a DataFrame for feature importance\nfeat_imp_df = pd.DataFrame({\n    'feature': feature_names,\n    'importance': importances\n}).sort_values(by='importance', ascending=False)\n\n# Show top N features\ntop_n = 10\ntop_features = feat_imp_df.head(top_n)\nprint(top_features)\n\n# Optional: plot\nplt.figure(figsize=(10, 5))\nplt.barh(top_features['feature'], top_features['importance'], color='skyblue')\nplt.xlabel(\"Feature Importance\")\nplt.title(\"Top 10 Features by Random Forest\")\nplt.gca().invert_yaxis()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:07:37.375474Z","iopub.execute_input":"2025-05-28T07:07:37.375894Z","iopub.status.idle":"2025-05-28T07:07:37.852900Z","shell.execute_reply.started":"2025-05-28T07:07:37.375874Z","shell.execute_reply":"2025-05-28T07:07:37.852149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}