{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81000,"databundleVersionId":8812083,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import re, gc\nimport pandas as pd\nimport numpy as np\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\npd.set_option('display.max_columns', None)\ngc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-14T20:27:25.205887Z","iopub.execute_input":"2024-07-14T20:27:25.206265Z","iopub.status.idle":"2024-07-14T20:27:26.421622Z","shell.execute_reply.started":"2024-07-14T20:27:25.206233Z","shell.execute_reply":"2024-07-14T20:27:26.420271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce memory usage of a pandas DataFrame\ndef reduce_memory_usage(df):\n    \"\"\"Reduce memory usage of a pandas DataFrame.\"\"\"\n    # Function to iterate through columns and modify the data types\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(f\"Memory usage of dataframe: {start_mem} MB\")\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == \"int\":\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f\"Memory usage after optimization: {end_mem} MB\")\n    print(f\"Decreased by {100 * (start_mem - end_mem) / start_mem}%\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-07-14T20:27:26.423953Z","iopub.execute_input":"2024-07-14T20:27:26.424562Z","iopub.status.idle":"2024-07-14T20:27:26.438268Z","shell.execute_reply.started":"2024-07-14T20:27:26.424517Z","shell.execute_reply":"2024-07-14T20:27:26.436986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Datasets\n## Train Set\n","metadata":{}},{"cell_type":"code","source":"# Define file paths and names for maize train datasets\nmaize_train_datasets = [\n    {'file_name': 'pr_', 'path': '/kaggle/input/the-future-crop-challenge/pr_maize_train.parquet'},\n    {'file_name': 'soil_co2_', 'path': '/kaggle/input/the-future-crop-challenge/soil_co2_maize_train.parquet'},\n    {'file_name': 'tas_', 'path': '/kaggle/input/the-future-crop-challenge/tas_maize_train.parquet'},\n    {'file_name': 'tasmin_', 'path': '/kaggle/input/the-future-crop-challenge/tasmin_maize_train.parquet'},\n    {'file_name': 'tasmax_', 'path': '/kaggle/input/the-future-crop-challenge/tasmax_maize_train.parquet'},\n    {'file_name': 'rsds_', 'path': '/kaggle/input/the-future-crop-challenge/rsds_maize_train.parquet'},\n    {'file_name': '', 'path': '/kaggle/input/the-future-crop-challenge/train_solutions_maize.parquet'}\n]\n\n# Read and store each train dataset with file name prefix\nmaize_train_dfs = [pd.read_parquet(file['path']).add_prefix(file['file_name']) for file in maize_train_datasets]\n\n# Concatenate all maize train datasets horizontally\nmaize_train_df = pd.concat(maize_train_dfs, axis=1)\nmaize_train_df = reduce_memory_usage(maize_train_df)\n\ndel maize_train_dfs\ngc.collect()\n\n# Define file paths and names for wheat train datasets\nwheat_train_datasets = [\n    {'file_name': 'pr_', 'path': '/kaggle/input/the-future-crop-challenge/pr_wheat_train.parquet'},\n    {'file_name': 'soil_co2_', 'path': '/kaggle/input/the-future-crop-challenge/soil_co2_wheat_train.parquet'},\n    {'file_name': 'tas_', 'path': '/kaggle/input/the-future-crop-challenge/tas_wheat_train.parquet'},\n    {'file_name': 'tasmin_', 'path': '/kaggle/input/the-future-crop-challenge/tasmin_wheat_train.parquet'},\n    {'file_name': 'tasmax_', 'path': '/kaggle/input/the-future-crop-challenge/tasmax_wheat_train.parquet'},\n    {'file_name': 'rsds_', 'path': '/kaggle/input/the-future-crop-challenge/rsds_wheat_train.parquet'},\n    {'file_name': '', 'path': '/kaggle/input/the-future-crop-challenge/train_solutions_wheat.parquet'}\n]\n\n# Read and store each wheat train dataset with file name prefix\nwheat_train_dfs = [pd.read_parquet(file['path']).add_prefix(file['file_name']) for file in wheat_train_datasets]\n\n# Concatenate all wheat train datasets horizontally\nwheat_train_df = pd.concat(wheat_train_dfs, axis=1)\nwheat_train_df = reduce_memory_usage(wheat_train_df)\n\ndel wheat_train_dfs\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-07-14T20:27:26.445905Z","iopub.execute_input":"2024-07-14T20:27:26.446336Z","iopub.status.idle":"2024-07-14T20:28:41.622014Z","shell.execute_reply.started":"2024-07-14T20:27:26.446296Z","shell.execute_reply":"2024-07-14T20:28:41.620832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([maize_train_df, wheat_train_df], ignore_index=False)\n\ndel maize_train_df, wheat_train_df\ngc.collect()\n\ncolumns_to_drop = [\n    'pr_crop', 'pr_year', 'pr_lon', 'pr_lat', 'pr_variable',\n    'tas_crop', 'tas_year', 'tas_lon', 'tas_lat', 'tas_variable',\n    'tasmin_crop', 'tasmin_year', 'tasmin_lon', 'tasmin_lat', 'tasmin_variable',\n    'tasmax_crop', 'tasmax_year', 'tasmax_lon', 'tasmax_lat', 'tasmax_variable',\n    'rsds_crop', 'rsds_year', 'rsds_lon', 'rsds_lat', 'rsds_variable'\n]\n\ntrain.drop(columns=columns_to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-07-14T20:28:41.623632Z","iopub.execute_input":"2024-07-14T20:28:41.624274Z","iopub.status.idle":"2024-07-14T20:28:45.849346Z","shell.execute_reply.started":"2024-07-14T20:28:41.624234Z","shell.execute_reply":"2024-07-14T20:28:45.848114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-14T20:28:45.850733Z","iopub.execute_input":"2024-07-14T20:28:45.851158Z","iopub.status.idle":"2024-07-14T20:28:46.718971Z","shell.execute_reply.started":"2024-07-14T20:28:45.851119Z","shell.execute_reply":"2024-07-14T20:28:46.717863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test set","metadata":{}},{"cell_type":"code","source":"# Define file paths and names for maize test datasets\nmaize_test_datasets = [\n    {'file_name': 'pr', 'path': '/kaggle/input/the-future-crop-challenge/pr_maize_test.parquet'},\n    {'file_name': 'soil_co2', 'path': '/kaggle/input/the-future-crop-challenge/soil_co2_maize_test.parquet'},\n    {'file_name': 'tas', 'path': '/kaggle/input/the-future-crop-challenge/tas_maize_test.parquet'},\n    {'file_name': 'tasmin', 'path': '/kaggle/input/the-future-crop-challenge/tasmin_maize_test.parquet'},\n    {'file_name': 'tasmax', 'path': '/kaggle/input/the-future-crop-challenge/tasmax_maize_test.parquet'},\n    {'file_name': 'rsds', 'path': '/kaggle/input/the-future-crop-challenge/rsds_maize_test.parquet'}\n]\n\n# Read and store each test dataset with file name prefix\nmaize_test_dfs = [pd.read_parquet(file['path']).add_prefix(file['file_name'] + '_') for file in maize_test_datasets]\n\n# Concatenate all maize test datasets horizontally\nmaize_test_df = pd.concat(maize_test_dfs, axis=1)\nmaize_test_df = reduce_memory_usage(maize_test_df)\n\ndel maize_test_dfs\ngc.collect()\n\n# Define file paths and names for wheat test datasets\nwheat_test_datasets = [\n    {'file_name': 'pr', 'path': '/kaggle/input/the-future-crop-challenge/pr_wheat_test.parquet'},\n    {'file_name': 'soil_co2', 'path': '/kaggle/input/the-future-crop-challenge/soil_co2_wheat_test.parquet'},\n    {'file_name': 'tas', 'path': '/kaggle/input/the-future-crop-challenge/tas_wheat_test.parquet'},\n    {'file_name': 'tasmin', 'path': '/kaggle/input/the-future-crop-challenge/tasmin_wheat_test.parquet'},\n    {'file_name': 'tasmax', 'path': '/kaggle/input/the-future-crop-challenge/tasmax_wheat_test.parquet'},\n    {'file_name': 'rsds', 'path': '/kaggle/input/the-future-crop-challenge/rsds_wheat_test.parquet'}\n]\n\n# Read and store each wheat test dataset with file name prefix\nwheat_test_dfs = [pd.read_parquet(file['path']).add_prefix(file['file_name'] + '_') for file in wheat_test_datasets]\n\n# Concatenate all wheat test datasets horizontally\nwheat_test_df = pd.concat(wheat_test_dfs, axis=1)\nwheat_test_df = reduce_memory_usage(wheat_test_df)\n\ndel wheat_test_dfs\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-07-14T20:28:46.720870Z","iopub.execute_input":"2024-07-14T20:28:46.721280Z","iopub.status.idle":"2024-07-14T20:31:21.559093Z","shell.execute_reply.started":"2024-07-14T20:28:46.721245Z","shell.execute_reply":"2024-07-14T20:31:21.557946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.concat([maize_test_df, wheat_test_df], ignore_index=False)\n\ndel maize_test_df, wheat_test_df\ngc.collect()\n\ncolumns_to_drop = [\n    'pr_crop', 'pr_year', 'pr_lon', 'pr_lat', 'pr_variable',\n    'tas_crop', 'tas_year', 'tas_lon', 'tas_lat', 'tas_variable',\n    'tasmin_crop', 'tasmin_year', 'tasmin_lon', 'tasmin_lat', 'tasmin_variable',\n    'tasmax_crop', 'tasmax_year', 'tasmax_lon', 'tasmax_lat', 'tasmax_variable',\n    'rsds_crop', 'rsds_year', 'rsds_lon', 'rsds_lat', 'rsds_variable'\n]\n\ntest.drop(columns=columns_to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-07-14T20:31:21.560649Z","iopub.execute_input":"2024-07-14T20:31:21.560993Z","iopub.status.idle":"2024-07-14T20:31:30.307677Z","shell.execute_reply.started":"2024-07-14T20:31:21.560963Z","shell.execute_reply":"2024-07-14T20:31:30.306350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-14T20:31:30.309034Z","iopub.execute_input":"2024-07-14T20:31:30.309385Z","iopub.status.idle":"2024-07-14T20:31:31.254694Z","shell.execute_reply.started":"2024-07-14T20:31:30.309355Z","shell.execute_reply":"2024-07-14T20:31:31.253518Z"},"trusted":true},"execution_count":null,"outputs":[]}]}