{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Reference Notebooks:**\n* [EDA+Google Smartphone Decimeter](https://www.kaggle.com/yasserhessein/eda-google-smartphone-decimeter)","metadata":{}},{"cell_type":"markdown","source":"**Import Required libraries**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Loading train, test & submission files**","metadata":{}},{"cell_type":"code","source":"trainfile = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_train.csv\")\ntestfile = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/baseline_locations_test.csv\")\nsubmission = pd.read_csv(\"../input/google-smartphone-decimeter-challenge/sample_submission.csv\")\n# submission = submission.assign(latDeg=test.latDeg.round(6), lngDeg=test.lngDeg.round(6))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainfile.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testfile.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainfile.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testfile.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Extracting ground truths and aligning with train inputs**","metadata":{}},{"cell_type":"code","source":"datapath = Path(\"../input/google-smartphone-decimeter-challenge\")\ntruths = (datapath / 'train').rglob('ground_truth.csv')\n\n\ncols = ['collectionName', 'phoneName', 'millisSinceGpsEpoch', 'latDeg',\n       'lngDeg']\ntruth_list =[]\nfor filepath in tqdm(truths, total=73):\n    file = pd.read_csv(filepath, usecols=cols)\n    truth_list.append(file)\n    \ntruth_data = pd.concat(truth_list, ignore_index=True)\n\ntrain = trainfile[cols]\n\ntrain = train.merge(truth_data, on=cols[:3], suffixes=(\"_current\",\"_truth\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Checking the correlation between current & truth coordinates**","metadata":{}},{"cell_type":"code","source":"train.iloc[:,3:].corr()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* In above correlation table, we can see that ***latDeg_current is highly correlated with latDeg_truth*** and ***lngDeg_current is highly correlated with lngDeg_truth***\n* That would help us to build a baseline model","metadata":{}},{"cell_type":"code","source":"test = testfile.copy()\n\nprint(\"############### collectionName unique values ##############################\")\nprint(\"train: {}\".format(train.collectionName.nunique()))\nprint(train.collectionName.unique())\nprint(\"----------------------------------------------\")\nprint(\"test: {}\".format(test.collectionName.nunique()))\nprint(test.collectionName.unique())\nprint(\"----------------------------------------------\")\n\nprint(\"\\n\")\n\nprint(\"############### phoneName unique values ##############################\")\nprint(\"train: {}\".format(train.phoneName.nunique()))\nprint(train.phoneName.unique())\nprint(\"----------------------------------------------\")\nprint(\"test: {}\".format(test.phoneName.nunique()))\nprint(test.phoneName.unique())\nprint(\"----------------------------------------------\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Note:** Here, collectionName would be useless to consider as it would be different in train and test data but **phoneName** might be helpful","metadata":{}},{"cell_type":"markdown","source":"**One-hot encoding for 'phoneName' of train and test data**","metadata":{}},{"cell_type":"code","source":"train_phone = pd.get_dummies(train.loc[:,\"phoneName\"])\ntest_phone = pd.get_dummies(test.loc[:,\"phoneName\"])\n\nprint(\"train_phone shape:{}\".format(train_phone.shape))\nprint(\"test_phone shape:{}\".format(test_phone.shape))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Aligning train & test columns**","metadata":{}},{"cell_type":"code","source":"train_phone1, test_phone1 = train_phone.align(test_phone, join=\"outer\",axis=1, fill_value=0)\nprint(\"Updated train shape {}\".format(train_phone.shape))\n\nprint(\"Updated test shape {}\".format(test_phone.shape))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Basically, there is no change is one-hot encoded phoneName columns for train or test data","metadata":{}},{"cell_type":"markdown","source":"**Train-test data after addition of one-hot encoded columns**","metadata":{}},{"cell_type":"code","source":"train1 = pd.concat([train.iloc[:,3:], train_phone1], axis=1, ignore_index=False)\ntest1 = pd.concat([test.iloc[:,3:5], test_phone1], axis=1, ignore_index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"train1_shape:\",train1.shape)\ntrain1.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"test1-shape:\",test1.shape)\ntest1.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plotting current vs truth coordinates**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=[10,5])\nplt.plot(train1[\"latDeg_current\"][:200],train1[\"lngDeg_current\"][:200],\"bo\",label=\"current\")\nplt.plot(train1[\"latDeg_truth\"][:200],train1[\"lngDeg_truth\"][:200],\"r*\",label=\"truth\")\nplt.title(\"current vs truth\", fontweight=\"bold\")\nplt.xlabel(\"latDeg\")\nplt.ylabel(\"lngDeg\")\nplt.legend()\n\nplt.figure(figsize=[15,5])\nplt.subplot(1,2,1)\nplt.plot(train1[\"latDeg_current\"][:200],train1[\"latDeg_truth\"][:200],\"bo\")\nplt.title(\"lat (current vs truth)\", fontweight=\"bold\")\n\nplt.subplot(1,2,2)\nplt.plot(train1[\"lngDeg_current\"][:200],train1[\"lngDeg_truth\"][:200],\"bo\")\nplt.title(\"lng (current vs truth)\", fontweight=\"bold\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Input-output**","metadata":{}},{"cell_type":"code","source":"# for lat\ncolumns1 = list(train1.columns[4:])\nX1,y1 = train1.loc[:,[train1.columns[0]]+columns1],train1[\"latDeg_truth\"].values\n  \n#for lng\ncolumns2 = list(train1.columns[4:])\nX2,y2 = train1.loc[:,[train1.columns[1]]+columns2],train1[\"lngDeg_truth\"].values\n\n\nXt1 = test1.loc[:,[test1.columns[0]]+columns1] #for lat\nXt2 = test1.loc[:,[test1.columns[1]]+columns2] #for lng\n\nprint(\"X1 columns:\", X1.columns.tolist())\nprint(\"Xt1 columns:\", Xt1.columns.tolist())\nprint(\"\\n\")\nprint(\"X2 columns:\", X2.columns.tolist())\nprint(\"Xt2 columns:\", Xt2.columns.tolist())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtr1,xval1,ytr1,yval1 = train_test_split(X1, y1, test_size=0.3, random_state=10)\nxtr2,xval2,ytr2,yval2 = train_test_split(X2, y2, test_size=0.3, random_state=10)\n\nprint(\"xtr1 shape:{}; xval1 shape:{}\".format(xtr1.shape,xval1.shape))\nprint(\"xtr2 shape:{}; xval2 shape:{}\".format(xtr2.shape,xval2.shape))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xval1.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xval2.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Defining a function to get prediction and ground truth distance estimation (in meters)**\n\n[Haversine formula for distance estimation using GPS co-ordinates](https://stackoverflow.com/questions/15736995/how-can-i-quickly-estimate-the-distance-between-two-latitude-longitude-points)","metadata":{}},{"cell_type":"code","source":"from math import radians, cos, sin, asin, sqrt\ndef lat_lon_dist(df):\n    \"\"\"\n    Calculate the great circle distance between two points \n    on the earth (specified in decimal degrees)\n    \"\"\"\n    dist_list = []\n    for i in tqdm(range(df.shape[0]),total=100):\n        lat1 = df[\"latDeg_truth\"][i]\n        lon1 = df[\"lngDeg_truth\"][i]\n        lat2 = df[\"latDeg_pred\"][i]\n        lon2 = df[\"lngDeg_pred\"][i]\n        # convert decimal degrees to radians \n        lon1, lat1, lon2, lat2 = map(radians, [lon1, lat1, lon2, lat2])\n        # haversine formula \n        dlon = lon2 - lon1 \n        dlat = lat2 - lat1 \n        a = sin(dlat/2)**2 + cos(lat1) * cos(lat2) * sin(dlon/2)**2\n        c = 2 * asin(sqrt(a)) \n        # Radius of earth in kilometers is 6371\n        mdist = 6371* c*1000\n        dist_list.append(mdist)\n    \n    return dist_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model building**","metadata":{}},{"cell_type":"code","source":"lr1 = LinearRegression() #selected as a starting point\nmodel_lat = lr1.fit(xtr1,ytr1)\npred_yval1 = model_lat.predict(xval1) # prediction for val data (lat)\n\n\nlr2 = LinearRegression()\nmodel_lng = lr2.fit(xtr2,ytr2)\npred_yval2 = model_lng.predict(xval2) # prediction for val data (long)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model evaluation on validation data**","metadata":{}},{"cell_type":"code","source":"val_df = pd.concat([xval1[[\"latDeg_current\"]], xval2], ignore_index=False, axis=1).reset_index(drop=[\"index\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Adding truth & predicted lat-long values to val_df**","metadata":{}},{"cell_type":"code","source":"#truth\nval_df[\"latDeg_truth\"] = yval1\nval_df[\"lngDeg_truth\"] = yval2\n\n#pred\nval_df[\"latDeg_pred\"] = pred_yval1\nval_df[\"lngDeg_pred\"] = pred_yval2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df[\"dist\"] = lat_lon_dist(val_df)\n\nval_df.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"phone = val_df.iloc[:,2:-5].idxmax(axis=1) #Reversing one-hot decoding for phoneName\n\nval_df1 = pd.concat([val_df.iloc[:,:2],val_df.iloc[:,-3:]], axis=1, ignore_index=False)\nval_df1[\"phoneName\"] = phone\n\nval_df1 = val_df1[val_df1.columns[-1:].tolist()+val_df1.columns[:-1].tolist()]\nval_df1.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Box-plot analysis for dist  analysis for each phone**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nplt.figure(figsize=[15,7])\n\n# ax, fig = plt.subplots(figsize=[15,7])\nsns.boxplot(x=\"phoneName\", y=\"dist\",data=val_df1)\nplt.ylabel(\"Dist (m)\") # distance in meters\nplt.ylim([0,30]) # for better visualization","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Preparing evaluation score for each phone (50th & 95th percentile)**","metadata":{}},{"cell_type":"code","source":"val_df2 = pd.DataFrame()\nval_df2[\"phoneName\"] =  val_df1.phoneName.unique().tolist()\nval_df2[\"dist_50\"] = [np.percentile(val_df1[val_df1.phoneName==ph][\"dist\"],50) for ph in val_df2[\"phoneName\"].tolist()]\nval_df2[\"dist_95\"] = [np.percentile(val_df1[val_df1.phoneName==ph][\"dist\"],95) for ph in val_df2[\"phoneName\"].tolist()]\nval_df2[\"avg_dist_50_95\"] = np.mean(np.array(val_df2.iloc[:,1:]),axis=1)\nprint(\"Val evaluation details:\\n\",val_df2)\n\nprint(\"\\n\")\nprint(\"------------------------------------------------------\")\nprint(\"Final val evaluation score: {}\".format(val_df2.iloc[:,-1].mean()))\nprint(\"------------------------------------------------------\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training the model on complete data and predict for test data**","metadata":{}},{"cell_type":"code","source":"lr1 = LinearRegression()\nmodel_lat = lr1.fit(X1,y1)\npred_yt1 = model_lat.predict(Xt1) # prediction for test data (lat)\n\nlr2 = LinearRegression()\nmodel_lng = lr2.fit(X2,y2)\npred_yt2 = model_lng.predict(Xt2) # prediction for test data (long)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = test[['phone','millisSinceGpsEpoch']]\nsubmission['latDeg'] = list(pred_yt1)\nsubmission['lngDeg'] = list(pred_yt2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"./submission.csv\",index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}