{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-14T04:17:34.311831Z","iopub.execute_input":"2022-08-14T04:17:34.312328Z","iopub.status.idle":"2022-08-14T04:17:34.323482Z","shell.execute_reply.started":"2022-08-14T04:17:34.312290Z","shell.execute_reply":"2022-08-14T04:17:34.322117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/tabular-playground-series-feb-2021/train.csv')\ntest =  pd.read_csv('/kaggle/input/tabular-playground-series-feb-2021/test.csv')\nsub =  pd.read_csv('/kaggle/input/tabular-playground-series-feb-2021/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:17:35.223416Z","iopub.execute_input":"2022-08-14T04:17:35.223871Z","iopub.status.idle":"2022-08-14T04:17:41.483273Z","shell.execute_reply.started":"2022-08-14T04:17:35.223835Z","shell.execute_reply":"2022-08-14T04:17:41.481987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for Null values\ntrain.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:17:41.486102Z","iopub.execute_input":"2022-08-14T04:17:41.486662Z","iopub.status.idle":"2022-08-14T04:17:41.633574Z","shell.execute_reply.started":"2022-08-14T04:17:41.486613Z","shell.execute_reply":"2022-08-14T04:17:41.632179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets check about datatype of all attributes\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:17:41.635428Z","iopub.execute_input":"2022-08-14T04:17:41.636329Z","iopub.status.idle":"2022-08-14T04:17:41.805207Z","shell.execute_reply.started":"2022-08-14T04:17:41.636278Z","shell.execute_reply":"2022-08-14T04:17:41.803673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Id column does not have much to contribute, so we'll drop it\ntrain.drop('id', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:17:41.808071Z","iopub.execute_input":"2022-08-14T04:17:41.808488Z","iopub.status.idle":"2022-08-14T04:17:41.893305Z","shell.execute_reply.started":"2022-08-14T04:17:41.808451Z","shell.execute_reply":"2022-08-14T04:17:41.891886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets get some training data\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:17:41.895366Z","iopub.execute_input":"2022-08-14T04:17:41.895868Z","iopub.status.idle":"2022-08-14T04:17:41.930571Z","shell.execute_reply.started":"2022-08-14T04:17:41.895818Z","shell.execute_reply":"2022-08-14T04:17:41.929473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Since, columns 1 to 10 are of dtype - object(A, B, C, D, E..), we'll use encoding to make them looklike the rest of the column values\n\ncategorical_cols=['cat'+str(i) for i in range(10)]\ncontinous_cols=['cont'+str(i) for i in range(14)]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:53.861628Z","iopub.execute_input":"2022-08-14T04:24:53.862148Z","iopub.status.idle":"2022-08-14T04:24:53.869842Z","shell.execute_reply.started":"2022-08-14T04:24:53.862112Z","shell.execute_reply":"2022-08-14T04:24:53.868416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfor e in categorical_cols:\n    le = LabelEncoder()\n    train[e]=le.fit_transform(train[e])\n    test[e]=le.transform(test[e])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:54.202308Z","iopub.execute_input":"2022-08-14T04:24:54.202757Z","iopub.status.idle":"2022-08-14T04:24:54.396168Z","shell.execute_reply.started":"2022-08-14T04:24:54.202722Z","shell.execute_reply":"2022-08-14T04:24:54.394859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=train[categorical_cols+continous_cols]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:55.087997Z","iopub.execute_input":"2022-08-14T04:24:55.088725Z","iopub.status.idle":"2022-08-14T04:24:55.135625Z","shell.execute_reply.started":"2022-08-14T04:24:55.088687Z","shell.execute_reply":"2022-08-14T04:24:55.134513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data\ny = train['target']","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:56.575073Z","iopub.execute_input":"2022-08-14T04:24:56.575566Z","iopub.status.idle":"2022-08-14T04:24:56.581959Z","shell.execute_reply.started":"2022-08-14T04:24:56.575529Z","shell.execute_reply":"2022-08-14T04:24:56.580836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:25:07.328830Z","iopub.execute_input":"2022-08-14T04:25:07.329338Z","iopub.status.idle":"2022-08-14T04:25:07.520651Z","shell.execute_reply.started":"2022-08-14T04:25:07.329302Z","shell.execute_reply":"2022-08-14T04:25:07.519157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xg\n\nxgb_reg = xg.XGBRegressor(objective ='reg:squarederror',n_estimators = 10, seed = 123)\nxgb_reg.fit(X_train, y_train)\n\nprediction = xgb_reg.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:28:48.885142Z","iopub.execute_input":"2022-08-14T04:28:48.885617Z","iopub.status.idle":"2022-08-14T04:28:54.040885Z","shell.execute_reply.started":"2022-08-14T04:28:48.885580Z","shell.execute_reply":"2022-08-14T04:28:54.039765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, mean_absolute_error\n\nmse = mean_squared_error(prediction, y_test)\nmae = mean_absolute_error(prediction, y_test)\n\nprint('mse - ', mse, 'mae - ',mae)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:29:50.447269Z","iopub.execute_input":"2022-08-14T04:29:50.447725Z","iopub.status.idle":"2022-08-14T04:29:50.459554Z","shell.execute_reply.started":"2022-08-14T04:29:50.447690Z","shell.execute_reply":"2022-08-14T04:29:50.458108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prediction = xgb_reg.predict(test.drop('id', axis = 1))\ntest_prediction\n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:38:14.561004Z","iopub.execute_input":"2022-08-14T04:38:14.561489Z","iopub.status.idle":"2022-08-14T04:38:14.665956Z","shell.execute_reply.started":"2022-08-14T04:38:14.561452Z","shell.execute_reply":"2022-08-14T04:38:14.664476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"--------------------------------------------------------------------------------","metadata":{}},{"cell_type":"code","source":"pred = xgb_reg.predict(test.drop('id', axis = 1))\noutput = pd.DataFrame({'id':test['id'],'target':pred})\noutput.set_index('id',inplace=True)\noutput.to_csv('output.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:37:38.678683Z","iopub.execute_input":"2022-08-14T04:37:38.679165Z","iopub.status.idle":"2022-08-14T04:37:39.184139Z","shell.execute_reply.started":"2022-08-14T04:37:38.679130Z","shell.execute_reply":"2022-08-14T04:37:39.181774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:37:40.147502Z","iopub.execute_input":"2022-08-14T04:37:40.148038Z","iopub.status.idle":"2022-08-14T04:37:40.161947Z","shell.execute_reply.started":"2022-08-14T04:37:40.148001Z","shell.execute_reply":"2022-08-14T04:37:40.160535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['target']=pred\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:33:09.531060Z","iopub.execute_input":"2022-08-10T05:33:09.531431Z","iopub.status.idle":"2022-08-10T05:33:09.947184Z","shell.execute_reply.started":"2022-08-10T05:33:09.531401Z","shell.execute_reply":"2022-08-10T05:33:09.946240Z"},"trusted":true},"execution_count":null,"outputs":[]}]}