{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-03T05:50:41.010940Z","iopub.execute_input":"2021-08-03T05:50:41.011682Z","iopub.status.idle":"2021-08-03T05:50:41.025555Z","shell.execute_reply.started":"2021-08-03T05:50:41.011583Z","shell.execute_reply":"2021-08-03T05:50:41.024722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing Libraries\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nimport pandas_profiling as pp","metadata":{"execution":{"iopub.status.busy":"2021-08-03T05:58:30.876332Z","iopub.execute_input":"2021-08-03T05:58:30.876745Z","iopub.status.idle":"2021-08-03T05:58:32.249655Z","shell.execute_reply.started":"2021-08-03T05:58:30.876696Z","shell.execute_reply":"2021-08-03T05:58:32.248651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.read_csv('/kaggle/input/iml2021summer/train.csv')\ntest= pd.read_csv('/kaggle/input/iml2021summer/test.csv')\nsample= pd.read_csv('/kaggle/input/iml2021summer/sample_submission.csv')\ndf.profile_report()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T05:58:40.006149Z","iopub.execute_input":"2021-08-03T05:58:40.006534Z","iopub.status.idle":"2021-08-03T05:59:02.219031Z","shell.execute_reply.started":"2021-08-03T05:58:40.006502Z","shell.execute_reply":"2021-08-03T05:59:02.216826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:06:19.196379Z","iopub.execute_input":"2021-08-03T06:06:19.196769Z","iopub.status.idle":"2021-08-03T06:06:19.242618Z","shell.execute_reply.started":"2021-08-03T06:06:19.196737Z","shell.execute_reply":"2021-08-03T06:06:19.241623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:06:27.946098Z","iopub.execute_input":"2021-08-03T06:06:27.946471Z","iopub.status.idle":"2021-08-03T06:06:27.965859Z","shell.execute_reply.started":"2021-08-03T06:06:27.946435Z","shell.execute_reply":"2021-08-03T06:06:27.964645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:06:43.130729Z","iopub.execute_input":"2021-08-03T06:06:43.131131Z","iopub.status.idle":"2021-08-03T06:06:43.151612Z","shell.execute_reply.started":"2021-08-03T06:06:43.131097Z","shell.execute_reply":"2021-08-03T06:06:43.150496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = (df.dtypes == 'object')\nobject_cols = list(s[s].index)\n\nprint(\"Categorical variables:\")\nprint(object_cols)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:07:52.525238Z","iopub.execute_input":"2021-08-03T06:07:52.525651Z","iopub.status.idle":"2021-08-03T06:07:52.533498Z","shell.execute_reply.started":"2021-08-03T06:07:52.525615Z","shell.execute_reply":"2021-08-03T06:07:52.532131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_X_train = df.copy()\nlabel_X_valid = test.copy()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:09:29.706742Z","iopub.execute_input":"2021-08-03T06:09:29.707126Z","iopub.status.idle":"2021-08-03T06:09:29.712338Z","shell.execute_reply.started":"2021-08-03T06:09:29.707096Z","shell.execute_reply":"2021-08-03T06:09:29.711266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nordinal_encoder = OrdinalEncoder()\nlabel_X_train[object_cols] = ordinal_encoder.fit_transform(df[object_cols])\nlabel_X_valid[object_cols] = ordinal_encoder.transform(test[object_cols])\n\nprint(\"MAE from Approach 2 (Ordinal Encoding):\") \n# print(score_dataset(label_X_train, label_X_valid, y_train, y_valid))\nprint()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:10:51.247990Z","iopub.execute_input":"2021-08-03T06:10:51.248344Z","iopub.status.idle":"2021-08-03T06:10:51.288599Z","shell.execute_reply.started":"2021-08-03T06:10:51.248315Z","shell.execute_reply":"2021-08-03T06:10:51.287462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    label_X_train.profile_report()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:12:13.034931Z","iopub.execute_input":"2021-08-03T06:12:13.035301Z","iopub.status.idle":"2021-08-03T06:12:39.505532Z","shell.execute_reply.started":"2021-08-03T06:12:13.035271Z","shell.execute_reply":"2021-08-03T06:12:39.504630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y= label_X_train['Demand']\nX= label_X_train.drop('Demand',axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:17:47.860586Z","iopub.execute_input":"2021-08-03T06:17:47.860936Z","iopub.status.idle":"2021-08-03T06:17:47.867337Z","shell.execute_reply.started":"2021-08-03T06:17:47.860907Z","shell.execute_reply":"2021-08-03T06:17:47.866376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y,test_size=0.2,random_state=0)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:27:25.906720Z","iopub.execute_input":"2021-08-03T06:27:25.907117Z","iopub.status.idle":"2021-08-03T06:27:25.915643Z","shell.execute_reply.started":"2021-08-03T06:27:25.907084Z","shell.execute_reply":"2021-08-03T06:27:25.914414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# check xgboos\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_absolute_error","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:27:26.764913Z","iopub.execute_input":"2021-08-03T06:27:26.765307Z","iopub.status.idle":"2021-08-03T06:27:26.769571Z","shell.execute_reply.started":"2021-08-03T06:27:26.765276Z","shell.execute_reply":"2021-08-03T06:27:26.768512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb= XGBRegressor(learning_rate=0.1)\nxgb.fit(X_train,y_train)\npredictions = xgb.predict(X_valid)\nprint(\"Mean Absolute Error: \" + str(mean_absolute_error(predictions, y_valid)))\n","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:36:00.627701Z","iopub.execute_input":"2021-08-03T06:36:00.628124Z","iopub.status.idle":"2021-08-03T06:36:01.007915Z","shell.execute_reply.started":"2021-08-03T06:36:00.628083Z","shell.execute_reply":"2021-08-03T06:36:01.006999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_X_valid.info()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:35:50.415204Z","iopub.execute_input":"2021-08-03T06:35:50.415566Z","iopub.status.idle":"2021-08-03T06:35:50.433765Z","shell.execute_reply.started":"2021-08-03T06:35:50.415537Z","shell.execute_reply":"2021-08-03T06:35:50.432379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction= xgb.predict(label_X_valid)\noutput = pd.DataFrame({'Id': test['Id'],'Demand': prediction})\noutput.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:36:24.474224Z","iopub.execute_input":"2021-08-03T06:36:24.474726Z","iopub.status.idle":"2021-08-03T06:36:24.496664Z","shell.execute_reply.started":"2021-08-03T06:36:24.474692Z","shell.execute_reply":"2021-08-03T06:36:24.495870Z"},"trusted":true},"execution_count":null,"outputs":[]}]}