{"cells":[{"metadata":{},"cell_type":"markdown","source":"### "},{"metadata":{},"cell_type":"markdown","source":"# Part I"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"#in order to reduce size of data, we only use random 100000 data of training data as whole data above.\nimport pandas as pd\nimport random\nfrom random import randint\n \noldf=open('/kaggle/input/expedia-hotel-recommendations/train.csv','r',encoding='UTF-8')\nnewf=open('new_meta.csv','w',encoding='UTF-8')\nn = 0\n# sample(x,y)函数的作用是从序列x中，随机选择y个不重复的元素\nresultList = random.sample(range(1,753407),100000)\nlines=oldf.readlines()\nnewf.write(lines[0])\nfor i in resultList:\n    newf.write(lines[i])\n    \noldf.close()\nnewf.close()\nmeta_data=pd.read_csv('new_meta.csv')\nmeta_data.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"print(len(meta_data))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"meta_data.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"meta_data.isnull().any()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#in order to use ML model, we drop the cloumns whose type is object\n#we also need to separate x and y \n\nY = meta_data['is_booking']\nX = meta_data.drop(['date_time','orig_destination_distance','srch_ci','srch_co','is_booking'],axis=1)\nX.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#divide train data into 2 parts: \"learning\" set and \"testing\" set\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size=0.15, random_state=1)\nprint(len(X_train))\nprint(len(X_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#first ML algorithm: RandomForestClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score,precision_score,recall_score,f1_score,confusion_matrix\nclf = RandomForestClassifier(n_estimators=80)\nclf.fit(X_train, Y_train)\ny_pred=clf.predict(X_test)\nacc=accuracy_score(Y_test,y_pred)\n#prec=precision_score(Y_test, y_pred,average='micro')\n#recall=recall_score(Y_test, y_pred)\n#f1_v=f1_score(Y_test, y_pred)\nprint(\"Accuracy:\" ,acc)\nprint( \"confusion_matrix\")\nprint( confusion_matrix(Y_test, y_pred))\nfrom sklearn.metrics import classification_report\nprint(classification_report(Y_test, y_pred))\nprint(clf.feature_importances_)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#second ML algorithm: GaussianNB()\nfrom sklearn.naive_bayes import GaussianNB\nmnb = GaussianNB()\nmnb.fit(X_train,Y_train) \ny_predict = mnb.predict(X_test)\nacc=accuracy_score(Y_test,y_predict)\n#prec=precision_score(Y_test, y_pred,average='micro')\n#recall=recall_score(Y_test, y_pred)\n#f1_v=f1_score(Y_test, y_pred)\nprint(\"Accuracy:\" ,acc)\nprint( \"confusion_matrix\")\nprint( confusion_matrix(Y_test, y_predict))\nfrom sklearn.metrics import classification_report\nprint(classification_report(Y_test, y_predict))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Third ML algorithm:Logistic\nfrom sklearn.linear_model import LogisticRegression\nlr = LogisticRegression()\nlr.fit(X_train, Y_train)\ny_predict = lr.predict(X_test)\nacc=accuracy_score(Y_test,y_predict)\n#prec=precision_score(Y_test, y_pred,average='micro')\n#recall=recall_score(Y_test, y_pred)\n#f1_v=f1_score(Y_test, y_pred)\nprint(\"Accuracy:\" ,acc)\nprint( \"confusion_matrix\")\nprint( confusion_matrix(Y_test, y_predict))\nfrom sklearn.metrics import classification_report\nprint(classification_report(Y_test, y_predict))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Forth ML algorithm:KNN\nfrom sklearn import neighbors\nknn = neighbors.KNeighborsClassifier()\nknn.fit(X_train,Y_train)\ny_predict = knn.predict(X_test)\nacc=accuracy_score(Y_test,y_predict)\n#prec=precision_score(Y_test, y_pred,average='micro')\n#recall=recall_score(Y_test, y_pred)\n#f1_v=f1_score(Y_test, y_pred)\nprint(\"Accuracy:\" ,acc)\nprint( \"confusion_matrix\")\nprint( confusion_matrix(Y_test, y_predict))\nfrom sklearn.metrics import classification_report\nprint(classification_report(Y_test, y_predict))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Part II"},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport random\nfrom random import randint\n \noldf=open('/kaggle/input/expedia-hotel-recommendations/test.csv','r',encoding='UTF-8')\nnewf=open('new_choose.csv','w',encoding='UTF-8')\nn = 0\n# sample(x,y)函数的作用是从序列x中，随机选择y个不重复的元素\nresultList = random.sample(range(1,75342),6000)\nlines=oldf.readlines()\nnewf.write(lines[0])\nfor i in resultList:\n    newf.write(lines[i])\n    \noldf.close()\nnewf.close()\nmeta_data=pd.read_csv('new_choose.csv')\nmeta_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"meta_data.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"meta_data.groupby('is_mobile').count()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### is_package(clicking) for each AB-group"},{"metadata":{"trusted":true},"cell_type":"code","source":"meta_data.groupby('is_mobile')['is_package'].mean()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It appears that there was slight decrease in mobile connection when the booking was generated compare to the control when user connected from others. But while we are certain of the difference in the data, how certain should we be that mobile connection will be worse in the future?"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Creating an list with bootstrapped means for each AB-group\nboot_1d = []\nfor i in range(1000):\n    boot_mean = meta_data.sample(frac = 1,replace = True).groupby('is_mobile')['is_package'].mean()\n    boot_1d.append(boot_mean)\n    \n# Transforming the list to a DataFrame\nboot_1d = pd.DataFrame(boot_1d)\n    \n# A Kernel Density Estimate plot of the bootstrap distributions\nboot_1d.plot(kind='density')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"boot_1d.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Adding a column with the % difference between the two AB-groups\nboot_1d['diff'] = (boot_1d[0] - boot_1d[1])/boot_1d[1]*100\n\n# Ploting the bootstrap % difference\nax = boot_1d['diff'].plot(kind='density')\nax.set_title('% difference in is_package between the two AB-groups')\n\n# Calculating the probability that 1-day retention is greater when the gate is at level 30\nprint('Probability that click/booking is worse when use mobile connection:',(boot_1d['diff'] > 0).mean())","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}