{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport math\n\nimport warnings\nwarnings.filterwarnings('ignore', category=DeprecationWarning)\nwarnings.filterwarnings('ignore', category=FutureWarning)\nwarnings.filterwarnings('ignore', category=RuntimeWarning)\nwarnings.filterwarnings('ignore', category=UserWarning)\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\nscoring = 'r2'\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import AdaBoostRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.tree import DecisionTreeRegressor","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-27T17:09:23.234622Z","iopub.execute_input":"2021-06-27T17:09:23.235046Z","iopub.status.idle":"2021-06-27T17:09:25.091652Z","shell.execute_reply.started":"2021-06-27T17:09:23.234956Z","shell.execute_reply":"2021-06-27T17:09:25.090755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:10:05.554696Z","iopub.execute_input":"2021-06-27T17:10:05.555059Z","iopub.status.idle":"2021-06-27T17:10:05.908182Z","shell.execute_reply.started":"2021-06-27T17:10:05.555025Z","shell.execute_reply":"2021-06-27T17:10:05.907185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_test.csv\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:10:15.928399Z","iopub.execute_input":"2021-06-27T17:10:15.928765Z","iopub.status.idle":"2021-06-27T17:10:16.163716Z","shell.execute_reply.started":"2021-06-27T17:10:15.928735Z","shell.execute_reply":"2021-06-27T17:10:16.162962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train['collectionName'].unique()))\nprint(train['collectionName'].unique())\n\nprint(len(test['collectionName'].unique()))\nprint(test['collectionName'].unique())","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:10:26.843022Z","iopub.execute_input":"2021-06-27T17:10:26.843593Z","iopub.status.idle":"2021-06-27T17:10:26.892128Z","shell.execute_reply.started":"2021-06-27T17:10:26.843558Z","shell.execute_reply":"2021-06-27T17:10:26.891359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train['phoneName'].unique()))\nprint(train['phoneName'].unique())\n\nprint(len(test['phoneName'].unique()))\nprint(test['phoneName'].unique())","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:10:35.757383Z","iopub.execute_input":"2021-06-27T17:10:35.757763Z","iopub.status.idle":"2021-06-27T17:10:35.797153Z","shell.execute_reply.started":"2021-06-27T17:10:35.757729Z","shell.execute_reply":"2021-06-27T17:10:35.796012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train\ntrain.drop(['collectionName','heightAboveWgs84EllipsoidM','phone'],axis=1,inplace=True)\n\nx1_train=train.iloc[:,:-2]\ny1_train=train['latDeg']\ntrain_phone = pd.get_dummies(train.loc[:,\"phoneName\"])\nx1_train = pd.concat([x1_train,train_phone], axis=1, ignore_index=False)\nx1_train.drop(['phoneName'],axis=1,inplace=True)\nx_train1, x_val1, y_train1, y_val1 = train_test_split(x1_train, y1_train, test_size=0.33, random_state=42)\nx_train1, x_val1, y_train1, y_val1 = x_train1.values, x_val1.values, y_train1.values, y_val1.values\n\nx2_train=train.iloc[:,:-2]\ny2_train=train['lngDeg']\ntrain_phone = pd.get_dummies(train.loc[:,\"phoneName\"])\nx2_train = pd.concat([x2_train,train_phone], axis=1, ignore_index=False)\nx2_train.drop(['phoneName'],axis=1,inplace=True)\nx_train2, x_val2, y_train2, y_val2 = train_test_split(x2_train, y2_train, test_size=0.33, random_state=42)\nx_train2, x_val2, y_train2, y_val2 = x_train2.values, x_val2.values, y_train2.values, y_val2.values\n\n#test\ntest.drop(['collectionName','heightAboveWgs84EllipsoidM','phone'],axis=1,inplace=True)\n\nx1_test = test.iloc[:,:-2]\ny1_test = test['latDeg']\ntest_phone = pd.get_dummies(test.loc[:,\"phoneName\"])\nx1_test = pd.concat([x1_test, test_phone], axis=1, ignore_index=False)\nx1_test.drop(['phoneName'],axis=1,inplace=True)\nx_test1, y_test1 = x1_test.values, y1_test.values\n\nx2_test = test.iloc[:,:-2]\ny2_test = test['lngDeg']\ntest_phone = pd.get_dummies(test.loc[:,\"phoneName\"])\nx2_test = pd.concat([x2_test,test_phone], axis=1, ignore_index=False)\nx2_test.drop(['phoneName'],axis=1,inplace=True)\nx_test2, y_test2 = x2_test.values, y2_test.values","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:10:47.633166Z","iopub.execute_input":"2021-06-27T17:10:47.633546Z","iopub.status.idle":"2021-06-27T17:10:47.783971Z","shell.execute_reply.started":"2021-06-27T17:10:47.633514Z","shell.execute_reply":"2021-06-27T17:10:47.782985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def regression_report(y_true, y_pred):\n  print('Mean Absolute Error    : ',mean_absolute_error(y_true, y_pred))\n  print('Mean Squared Error     : ',mean_squared_error(y_true, y_pred))\n  print('Root Mean Squared Error: ',math.sqrt(mean_squared_error(y_true, y_pred)))\n  print('R-Squared              : ',r2_score(y_true, y_pred))\n\ndef regression(x_train, y_train, x_val, y_val):\n  #linear regression\n  reg1= LinearRegression()\n  print(\"                       Linear Regression\")\n  reg1.fit(x_train,y_train)\n  test_pred=reg1.predict(x_val)\n  print('Regression Report - Validation: ')\n  regression_report(y_val, test_pred)\n  print('\\n')\n\n  #Random Forest Regression\n  reg2= RandomForestRegressor(n_estimators=100)\n  print(\"                       Random Forest Regression: \")\n  reg2.fit(x_train,y_train)\n  test_pred=reg2.predict(x_val)\n  print('Regression Report - Validation: ')\n  regression_report(y_val, test_pred)\n  print('\\n')\n\n  #XGBoost Regression\n  reg3= XGBRegressor(verbosity = 0)\n  print(\"                       XGBoost Regression: \")\n  reg3.fit(x_train,y_train)\n  test_pred=reg3.predict(x_val)\n  print('Regression Report - Validation: ')\n  print(regression_report(y_val, test_pred))\n  print(\"\\n\")\n\n  #AdaBoost Regression\n  reg4= AdaBoostRegressor(n_estimators=100)\n  print(\"                       AdaBoost Regression: \")\n  reg4.fit(x_train,y_train)\n  test_pred=reg4.predict(x_val)\n  print('Regression Report - Validation: ')\n  regression_report(y_val, test_pred)\n  print(\"\\n\")\n\n  #KNeighbors Regression\n  reg5= KNeighborsRegressor()\n  print(\"                       KNeighbors Regression: \")\n  reg5.fit(x_train,y_train)\n  test_pred=reg5.predict(x_val)\n  print('Regression Report - Validation: ')\n  regression_report(y_val, test_pred)\n  print(\"\\n\")\n\n  #Decision Tree Regression\n  reg6= DecisionTreeRegressor()\n  print(\"                       Decision Tree Regression: \")\n  reg6.fit(x_train,y_train)\n  test_pred=reg6.predict(x_val)\n  print('Regression Report - Validation: ')\n  regression_report(y_val, test_pred)\n  print(\"\\n\")\n\nprint('                               latDeg')\nregression(x_train=x_train1, y_train=y_train1, x_val=x_val1, y_val=y_val1)\nprint('                               lngDeg')\nregression(x_train=x_train2, y_train=y_train2, x_val=x_val2, y_val=y_val2)","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:11:01.409453Z","iopub.execute_input":"2021-06-27T17:11:01.409851Z","iopub.status.idle":"2021-06-27T17:11:20.837536Z","shell.execute_reply.started":"2021-06-27T17:11:01.409817Z","shell.execute_reply":"2021-06-27T17:11:20.836455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reg = LinearRegression()\nreg.fit(x1_train,y1_train)\nlatDeg = reg.predict(x_test1)\n\nreg = LinearRegression()\nreg.fit(x2_train,y2_train)\nlngDeg = reg.predict(x_test2)","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:11:20.839940Z","iopub.execute_input":"2021-06-27T17:11:20.840422Z","iopub.status.idle":"2021-06-27T17:11:20.913790Z","shell.execute_reply.started":"2021-06-27T17:11:20.840373Z","shell.execute_reply":"2021-06-27T17:11:20.912280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.read_csv(\"../input/google-smartphone-decimeter-challenge/sample_submission.csv\")\nsubmission['latDeg']=latDeg\nsubmission['lngDeg']=lngDeg\nsubmission.to_csv(\"Submission.csv\",index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:18:09.813608Z","iopub.execute_input":"2021-06-27T17:18:09.813991Z","iopub.status.idle":"2021-06-27T17:18:10.663570Z","shell.execute_reply.started":"2021-06-27T17:18:09.813954Z","shell.execute_reply":"2021-06-27T17:18:10.662717Z"},"trusted":true},"execution_count":null,"outputs":[]}]}