{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"},{"sourceId":4474043,"sourceType":"datasetVersion","datasetId":2601572},{"sourceId":7595263,"sourceType":"datasetVersion","datasetId":1004280}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:14.617717Z","iopub.execute_input":"2024-12-17T14:48:14.618169Z","iopub.status.idle":"2024-12-17T14:48:14.642223Z","shell.execute_reply.started":"2024-12-17T14:48:14.618130Z","shell.execute_reply":"2024-12-17T14:48:14.641016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics.pairwise import pairwise_distances\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:14.644158Z","iopub.execute_input":"2024-12-17T14:48:14.644502Z","iopub.status.idle":"2024-12-17T14:48:15.552374Z","shell.execute_reply.started":"2024-12-17T14:48:14.644468Z","shell.execute_reply":"2024-12-17T14:48:15.550899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_book = pd.read_csv('/kaggle/input/book-recommendation-dataset/Books.csv')\ndf_rating = pd.read_csv('/kaggle/input/book-recommendation-dataset/Ratings.csv')\ndf_user = pd.read_csv('/kaggle/input/book-recommendation-dataset/Users.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:15.553851Z","iopub.execute_input":"2024-12-17T14:48:15.554466Z","iopub.status.idle":"2024-12-17T14:48:19.380155Z","shell.execute_reply.started":"2024-12-17T14:48:15.554426Z","shell.execute_reply":"2024-12-17T14:48:19.378779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_book.shape, df_rating.shape,df_user.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:19.382318Z","iopub.execute_input":"2024-12-17T14:48:19.382677Z","iopub.status.idle":"2024-12-17T14:48:19.390300Z","shell.execute_reply.started":"2024-12-17T14:48:19.382641Z","shell.execute_reply":"2024-12-17T14:48:19.389027Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Problem phrasing","metadata":{}},{"cell_type":"markdown","source":"* Based on the datasets provided, we will rely on closest neighborhood which is base on users' similarity to provide recommendation on new books\n* Objective: predict top 5 popular book for each individual\n* Performance measurement: if we have a perfect dataset in which everyone read the same books, then we can conduct cross-validation for hold-out books\n     ","metadata":{}},{"cell_type":"markdown","source":"# 1. Data preparation","metadata":{}},{"cell_type":"code","source":"# Dataset merge and treat missing, zero and extreme value\ndf_rating_user = pd.merge(df_rating,df_user,on = 'User-ID', how = 'left')\ndf_overall = pd.merge(df_rating_user,df_book,on = 'ISBN',how='left')\n\ndf_overall_v1 = df_overall.dropna()\n\nage_cap = df_overall_v1['Age'].quantile(0.99)\ndf_overall_vf = df_overall_v1[df_overall_v1['Age']<= age_cap]\n\ndf_overall_vf['loc_ctry'] = df_overall_vf['Location'].str.rsplit(',', n=1).str[-1]\n\n# KNN model treatment - numerical variables\nmin = df_overall_vf['Age'].min()\nmax = df_overall_vf['Age'].max()\ndf_overall_vf['Age_var'] = (df_overall_vf['Age']-min)/(max - min)\n\n# KNN model treatment - categorical variables\nfrom sklearn.preprocessing import OneHotEncoder\nX_encoder = OneHotEncoder(drop = 'first',sparse_output = False)\nX_encoded = X_encoder.fit_transform(df_overall_vf[['loc_ctry']])\nloc_name = [col + '_var' for col in X_encoder.get_feature_names_out()]\n\nX_encoded = pd.DataFrame(data=X_encoded,columns = loc_name)\ndf_overall_vf_v = df_overall_vf.reset_index() #for the purpose of pd.concat\ndf_overall_vf_v = pd.concat([df_overall_vf_v,X_encoded], axis = 1)\ndf_overall_vf_v.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:19.391547Z","iopub.execute_input":"2024-12-17T14:48:19.391878Z","iopub.status.idle":"2024-12-17T14:48:35.776794Z","shell.execute_reply.started":"2024-12-17T14:48:19.391845Z","shell.execute_reply":"2024-12-17T14:48:35.775675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"var_model = df_overall_vf_v.filter(like = '_var').columns\ndf_user_knn = df_overall_vf_v[var_model]\ndf_user_knn = pd.concat([df_overall_vf_v['User-ID'],df_user_knn],axis =1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:35.778276Z","iopub.execute_input":"2024-12-17T14:48:35.778619Z","iopub.status.idle":"2024-12-17T14:48:40.280342Z","shell.execute_reply.started":"2024-12-17T14:48:35.778584Z","shell.execute_reply":"2024-12-17T14:48:40.278968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_user_knn_v1 = df_user_knn.drop_duplicates()\ndf_user_knn_vf = df_user_knn_v1.drop(columns = 'User-ID')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:40.281978Z","iopub.execute_input":"2024-12-17T14:48:40.282468Z","iopub.status.idle":"2024-12-17T14:48:42.792206Z","shell.execute_reply.started":"2024-12-17T14:48:40.282419Z","shell.execute_reply":"2024-12-17T14:48:42.791164Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Data Feature Understanding","metadata":{}},{"cell_type":"markdown","source":"Given we plan to use users' age and location as dimension for similary,we use top2 popular book to demonstrate if there is homogeneity exsit based on these two features, and heterogeneity exist across the range of these two features","metadata":{}},{"cell_type":"code","source":"rating_summary = df_overall_vf.groupby('ISBN')['Book-Rating'].agg(['count']).reset_index()\nrating_summary_sort = rating_summary.sort_values(by ='count', ascending = False).head(2)\ntop_rated_book = pd.merge(rating_summary_sort,df_book, on = 'ISBN', how = 'left')\ntop_rated_book","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:42.793580Z","iopub.execute_input":"2024-12-17T14:48:42.793898Z","iopub.status.idle":"2024-12-17T14:48:44.132192Z","shell.execute_reply.started":"2024-12-17T14:48:42.793867Z","shell.execute_reply":"2024-12-17T14:48:44.127177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pull ratings for top 2 books to see if there is strong evidence that age and location determine the rating\ndf_overall_vf['age_bin'] = pd.qcut(df_overall_vf['Age'],10,labels = False)\n\ndf_rating_top1book = df_overall_vf[df_overall_vf['ISBN'] == '0971880107']\ndf_rating_top2book = df_overall_vf[df_overall_vf['ISBN'] == '0316666343']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:44.136480Z","iopub.execute_input":"2024-12-17T14:48:44.137705Z","iopub.status.idle":"2024-12-17T14:48:44.390188Z","shell.execute_reply.started":"2024-12-17T14:48:44.137516Z","shell.execute_reply":"2024-12-17T14:48:44.388982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"summary = df_rating_top1book.groupby(['Book-Rating','age_bin'])['Age'].count().reset_index().rename(columns = {\"Age\": \"Count\"})\nsns.barplot(summary, x = 'Book-Rating', y = 'Count', hue = 'age_bin')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:44.395432Z","iopub.execute_input":"2024-12-17T14:48:44.395932Z","iopub.status.idle":"2024-12-17T14:48:45.557043Z","shell.execute_reply.started":"2024-12-17T14:48:44.395867Z","shell.execute_reply":"2024-12-17T14:48:45.555564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_rating_top1book.groupby('loc_ctry')['Book-Rating'].agg(['count','mean','var','min','max']).reset_index().sort_values(by = 'count', ascending = False).head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:45.558778Z","iopub.execute_input":"2024-12-17T14:48:45.559264Z","iopub.status.idle":"2024-12-17T14:48:45.593959Z","shell.execute_reply.started":"2024-12-17T14:48:45.559213Z","shell.execute_reply":"2024-12-17T14:48:45.592446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"summary = df_rating_top2book.groupby(['Book-Rating','age_bin'])['Age'].count().reset_index().rename(columns = {\"Age\": \"Count\"})\nsns.barplot(summary, x = 'Book-Rating', y = 'Count', hue = 'age_bin')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:45.595775Z","iopub.execute_input":"2024-12-17T14:48:45.596442Z","iopub.status.idle":"2024-12-17T14:48:46.645605Z","shell.execute_reply.started":"2024-12-17T14:48:45.596390Z","shell.execute_reply":"2024-12-17T14:48:46.644361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_rating_top2book.groupby('loc_ctry')['Book-Rating'].agg(['count','mean','var','min','max']).reset_index().sort_values(by = 'count', ascending = False).head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:46.647045Z","iopub.execute_input":"2024-12-17T14:48:46.647389Z","iopub.status.idle":"2024-12-17T14:48:46.667320Z","shell.execute_reply.started":"2024-12-17T14:48:46.647356Z","shell.execute_reply":"2024-12-17T14:48:46.665999Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Create a recommendation book list for one top reader as a demo","metadata":{}},{"cell_type":"markdown","source":"Assuming there is homogeneity exist based on reader's profile, we first test out to create a recommendation book list based on the closest neighborhood for one top rater before apply to overall population.","metadata":{}},{"cell_type":"code","source":"# create a datasets for top 500 rater only,on avg they rate more than 200 books per person\ntt = df_overall_vf.groupby('User-ID')['Age'].count().reset_index().rename(columns = {\"Age\":\"Count\"}).sort_values(by='Count',ascending = False).head(500)\ndf_top_rator = pd.merge(tt,df_user_knn_v1,on = 'User-ID',how = 'left')\ndf_top_rator_knn = df_top_rator.drop(columns = ['User-ID','Count'])\n\ndistance_matrix = pairwise_distances(df_top_rator_knn, metric='euclidean')\ndf_distance = pd.DataFrame(distance_matrix)\ntt1 = tt.reset_index()\ntt1.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:46.668816Z","iopub.execute_input":"2024-12-17T14:48:46.669256Z","iopub.status.idle":"2024-12-17T14:48:46.756197Z","shell.execute_reply.started":"2024-12-17T14:48:46.669220Z","shell.execute_reply":"2024-12-17T14:48:46.753698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#use top1 rater, to calculate distance for all his/her neighborhood, to determine the best neighborhood size\ntop_rator_distance = pd.concat([tt1,df_distance],axis = 1)\ntop1_distance = top_rator_distance[top_rator_distance['User-ID']==198711]\ntop1_distance","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:46.757715Z","iopub.execute_input":"2024-12-17T14:48:46.758225Z","iopub.status.idle":"2024-12-17T14:48:46.811207Z","shell.execute_reply.started":"2024-12-17T14:48:46.758171Z","shell.execute_reply":"2024-12-17T14:48:46.810030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# rank order his/her neighborhood based on distance\ntd  = top1_distance.loc[:,0:]\ntd_transpose = td.transpose().rename(columns = {0:\"dis\"}).sort_values(by='dis',ascending = True)\ntop1_dis =pd.concat([tt1,td_transpose],axis = 1)\ntd_transpose_sort = top1_dis.sort_values(by = 'dis',ascending = True)\n\ntop1_neighbor = pd.merge(td_transpose_sort,df_user,on = 'User-ID', how = 'left')\ntop1_neighbor_rating = pd.merge(td_transpose_sort,df_rating_user,on = 'User-ID', how = 'left')\n\ntop1_neighbor.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:46.812556Z","iopub.execute_input":"2024-12-17T14:48:46.813037Z","iopub.status.idle":"2024-12-17T14:48:47.034763Z","shell.execute_reply.started":"2024-12-17T14:48:46.812976Z","shell.execute_reply":"2024-12-17T14:48:47.033521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#look at the similarity of top1 rater with his closest neighborhood\n#top4 circle\ntop1_circle4 = top1_neighbor_rating[top1_neighbor_rating['User-ID'].isin(top1_neighbor.head(4)['User-ID'])]\ntop1_circle4.groupby('ISBN')['Book-Rating'].agg(['count','min','max','mean','var']).reset_index().sort_values(by = 'count', ascending = False).head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:47.036306Z","iopub.execute_input":"2024-12-17T14:48:47.036673Z","iopub.status.idle":"2024-12-17T14:48:47.074305Z","shell.execute_reply.started":"2024-12-17T14:48:47.036637Z","shell.execute_reply":"2024-12-17T14:48:47.072990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#top20- circle\ntop1_circle20 = top1_neighbor_rating[top1_neighbor_rating['User-ID'].isin(top1_neighbor.head(20)['User-ID'])]\ntop1_circle20_rating = top1_circle20.groupby('ISBN')['Book-Rating'].agg(['count','min','max','mean','var']).reset_index().sort_values(by = 'count', ascending = False)\ntop1_circle20_rating[top1_circle20_rating['ISBN'].isin(['051513287X','0440211727','0440214041','0671886665'])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:47.075857Z","iopub.execute_input":"2024-12-17T14:48:47.076330Z","iopub.status.idle":"2024-12-17T14:48:47.134017Z","shell.execute_reply.started":"2024-12-17T14:48:47.076279Z","shell.execute_reply":"2024-12-17T14:48:47.132952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#top50- circle\ntop1_circle50 = top1_neighbor_rating[top1_neighbor_rating['User-ID'].isin(top1_neighbor.head(50)['User-ID'])]\ntop1_circle50_rating = top1_circle50.groupby('ISBN')['Book-Rating'].agg(['count','min','max','mean','var']).reset_index().sort_values(by = 'count', ascending = False)\ntop1_circle50_rating[top1_circle50_rating['ISBN'].isin(['051513287X','0440211727','0440214041','0671886665'])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:47.135616Z","iopub.execute_input":"2024-12-17T14:48:47.136109Z","iopub.status.idle":"2024-12-17T14:48:47.249711Z","shell.execute_reply.started":"2024-12-17T14:48:47.136061Z","shell.execute_reply":"2024-12-17T14:48:47.248169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#top500 circle\ntop1_circle500 = top1_neighbor_rating.groupby('ISBN')['Book-Rating'].agg(['count','min','max','mean','var']).reset_index().sort_values(by = 'count', ascending = False)\ntop1_circle500[top1_circle500['ISBN'].isin(['051513287X','0440211727','0440214041','0671886665'])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:47.251265Z","iopub.execute_input":"2024-12-17T14:48:47.251717Z","iopub.status.idle":"2024-12-17T14:48:47.674984Z","shell.execute_reply.started":"2024-12-17T14:48:47.251677Z","shell.execute_reply":"2024-12-17T14:48:47.674003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ee = []\nfor i in range(100):\n  top1_circle = top1_neighbor_rating[top1_neighbor_rating['User-ID'].isin(top1_neighbor.head(i)['User-ID'])]\n  top1_circle_rating = top1_circle.groupby('ISBN')['Book-Rating'].agg(['count','min','max','mean','var']).reset_index().sort_values(by = 'count', ascending = False)\n  dd = top1_circle_rating[top1_circle_rating['ISBN'].isin(['051513287X','0440211727','0440214041','0671886665'])]\n  ee.append(dd['var'].sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:47.676510Z","iopub.execute_input":"2024-12-17T14:48:47.676885Z","iopub.status.idle":"2024-12-17T14:48:56.281666Z","shell.execute_reply.started":"2024-12-17T14:48:47.676838Z","shell.execute_reply":"2024-12-17T14:48:56.280467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"i = np.arange(0,100,1)\nee = np.array(ee)\nsns.scatterplot(x = i, y = ee)\nplt.title('variance of book rating based on neighborhood size ')\nplt.xlabel('neighborhood size')\nplt.ylabel('variance of rating')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:56.283171Z","iopub.execute_input":"2024-12-17T14:48:56.283515Z","iopub.status.idle":"2024-12-17T14:48:56.593786Z","shell.execute_reply.started":"2024-12-17T14:48:56.283480Z","shell.execute_reply":"2024-12-17T14:48:56.592629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Based on homogeneity across neighborhood size, we make a call regarding neighborhood size","metadata":{}},{"cell_type":"code","source":"# find the book list that top1 rator has not read before\ncircle20_book_list = top1_circle20['ISBN'].unique()\ntop1_book_list = top1_circle20[top1_circle20['User-ID'] == top1_neighbor.head(1)['User-ID'][0]]['ISBN'].unique()\ntop1_non_read_list = set(circle20_book_list) - set(top1_book_list)\ntop1_non_read_list = pd.DataFrame(top1_non_read_list).rename(columns = {0:'ISBN'})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:56.595279Z","iopub.execute_input":"2024-12-17T14:48:56.595631Z","iopub.status.idle":"2024-12-17T14:48:56.616542Z","shell.execute_reply.started":"2024-12-17T14:48:56.595597Z","shell.execute_reply":"2024-12-17T14:48:56.615337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top1_non_read_circle_onebook_rating = top1_circle20[top1_circle20['ISBN'] == top1_non_read_list.iloc[2,0]]\ntop1_non_read_circle_onebook_rating.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:56.618159Z","iopub.execute_input":"2024-12-17T14:48:56.618518Z","iopub.status.idle":"2024-12-17T14:48:56.636785Z","shell.execute_reply.started":"2024-12-17T14:48:56.618485Z","shell.execute_reply":"2024-12-17T14:48:56.635199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top1_non_read_list['recommended_rating'] = 0\ntop1_non_read_list['recommended_rating_weightedvote'] = 0\ntop1_non_read_list['recommended_rating_vote'] = 0\n\nfor i in range(len(top1_non_read_list)):\n    top1_non_read_circle_onebook_rating = top1_circle20[top1_circle20['ISBN'] == top1_non_read_list.iloc[i,0]]\n    recommended = top1_non_read_circle_onebook_rating.groupby('Book-Rating')['dis'].agg(['sum','count']).reset_index().sort_values(by = 'sum', ascending = False)\n    recommanded_rating = recommended.iloc[0,0]\n    recommended_sum = recommended.iloc[0,1]\n    recommended_count = recommended.iloc[0,2]\n    \n    top1_non_read_list.iloc[i,1]=recommanded_rating\n    top1_non_read_list.iloc[i,2]=recommended_sum\n    top1_non_read_list.iloc[i,3]=recommended_count\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:48:56.638730Z","iopub.execute_input":"2024-12-17T14:48:56.639193Z","iopub.status.idle":"2024-12-17T14:50:14.927155Z","shell.execute_reply.started":"2024-12-17T14:48:56.639147Z","shell.execute_reply":"2024-12-17T14:50:14.925933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# propose the final recommended book, with highest rating and more than 1 \ntop1_non_read_list_sorted = top1_non_read_list.sort_values(by='recommended_rating',ascending = False)\nfinal_recommend_book = top1_non_read_list_sorted[top1_non_read_list_sorted['recommended_rating_vote']>0.05*20].head(5)\n\npd.merge(final_recommend_book,df_book,on= 'ISBN',how='left')[['recommended_rating','ISBN','Book-Title','Book-Author','Year-Of-Publication']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:50:14.928769Z","iopub.execute_input":"2024-12-17T14:50:14.929232Z","iopub.status.idle":"2024-12-17T14:50:15.078542Z","shell.execute_reply.started":"2024-12-17T14:50:14.929183Z","shell.execute_reply":"2024-12-17T14:50:15.077308Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Apply KNN model to overall population","metadata":{}},{"cell_type":"code","source":"tt = df_overall_vf.groupby('User-ID')['Age'].count().reset_index().rename(columns = {\"Age\":\"Count\"}).sort_values(by='Count',ascending = False).head(500)\ndf_top_rator = pd.merge(tt,df_user_knn_v1,on = 'User-ID',how = 'left')\ndf_top_rator_knn = df_top_rator.drop(columns = ['User-ID','Count'])\n\ndistance_matrix = pairwise_distances(df_top_rator_knn, metric='euclidean')\ndf_distance = pd.DataFrame(distance_matrix)\ntt1 = tt.reset_index()\ntop_rator_distance = pd.concat([tt1,df_distance],axis = 1)\naa = []\n\nfor i in range(10):\n    top1_distance = top_rator_distance[top_rator_distance['User-ID']==top_rator_distance['User-ID'][i]]\n    td  = top1_distance.loc[:,0:]\n    td_transpose = td.transpose().rename(columns = {i:\"dis\"}).sort_values(by='dis',ascending = True)\n    top1_dis =pd.concat([tt1,td_transpose],axis = 1)\n    td_transpose_sort = top1_dis.sort_values(by = 'dis',ascending = True)\n\n    top1_neighbor = pd.merge(td_transpose_sort,df_user,on = 'User-ID', how = 'left')\n    top1_neighbor_rating = pd.merge(td_transpose_sort,df_rating_user,on = 'User-ID', how = 'left')\n\n    #identify cloest neighborhood\n    top1_circle20 = top1_neighbor_rating[top1_neighbor_rating['User-ID'].isin(top1_neighbor.head(20)['User-ID'])]\n    circle20_book_list = top1_circle20['ISBN'].unique()\n    \n    #identify books have not been read by the individual\n    top1_book_list = top1_circle20[top1_circle20['User-ID'] == top1_neighbor.head(1)['User-ID'][0]]['ISBN'].unique()\n    top1_non_read_list = set(circle20_book_list) - set(top1_book_list)\n    top1_non_read_list = pd.DataFrame(top1_non_read_list).rename(columns = {0:'ISBN'})\n\n    #create a recommended rating for all unread books\n    top1_non_read_list['recommended_rating'] = 0\n    top1_non_read_list['recommended_rating_weightedvote'] = 0\n    top1_non_read_list['recommended_rating_vote'] = 0\n\n    for i in range(len(top1_non_read_list)):\n        top1_non_read_circle_onebook_rating = top1_circle20[top1_circle20['ISBN'] == top1_non_read_list.iloc[i,0]]\n        recommended = top1_non_read_circle_onebook_rating.groupby('Book-Rating')['dis'].agg(['sum','count']).reset_index().sort_values(by = 'sum', ascending = False)\n        recommanded_rating = recommended.iloc[0,0]\n        recommended_sum = recommended.iloc[0,1]\n        recommended_count = recommended.iloc[0,2]\n    \n        top1_non_read_list.iloc[i,1]=recommanded_rating\n        top1_non_read_list.iloc[i,2]=recommended_sum\n        top1_non_read_list.iloc[i,3]=recommended_count\n\n    #only recommend the ones with highest rating and with enough votes above threshold\n    top1_non_read_list_sorted = top1_non_read_list.sort_values(by='recommended_rating',ascending = False)\n    final_recommend_book = top1_non_read_list_sorted[top1_non_read_list_sorted['recommended_rating_vote']>0.05*20].head(5)\n    recommended_books = pd.merge(final_recommend_book,df_book,on= 'ISBN',how='left')[['recommended_rating','ISBN','Book-Title','Book-Author','Year-Of-Publication']]\n    aa.append(recommended_books)\n\nfor i in range(10):\n   aa[i]['id'] = df_top_rator['User-ID'][i]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:36:58.288555Z","iopub.execute_input":"2024-12-17T16:36:58.289033Z","iopub.status.idle":"2024-12-17T16:52:33.340428Z","shell.execute_reply.started":"2024-12-17T16:36:58.288990Z","shell.execute_reply":"2024-12-17T16:52:33.339366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# User enters a User-ID, and function will return a list of recommended book\n\nUser-ID = 76352\n\ndef book_recommendation (ID): \n    index = df_top_rator[df_top_rator['User-ID']==ID].index.tolist()[0]\n    return aa[index]\n\nbook_recommendation(User-ID)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:53:55.664892Z","iopub.execute_input":"2024-12-17T16:53:55.665991Z","iopub.status.idle":"2024-12-17T16:53:55.679727Z","shell.execute_reply.started":"2024-12-17T16:53:55.665901Z","shell.execute_reply":"2024-12-17T16:53:55.678458Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Word2vec e-commerce recommendation","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nfrom gensim.models import Word2Vec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:53:08.578947Z","iopub.status.idle":"2024-12-17T14:53:08.579352Z","shell.execute_reply.started":"2024-12-17T14:53:08.579183Z","shell.execute_reply":"2024-12-17T14:53:08.579203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pl.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pl.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:53:08.580883Z","iopub.status.idle":"2024-12-17T14:53:08.581290Z","shell.execute_reply.started":"2024-12-17T14:53:08.581124Z","shell.execute_reply":"2024-12-17T14:53:08.581143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.glimpse, test.glimpse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:53:08.583142Z","iopub.status.idle":"2024-12-17T14:53:08.583511Z","shell.execute_reply.started":"2024-12-17T14:53:08.583340Z","shell.execute_reply":"2024-12-17T14:53:08.583358Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Problem Phrasing\nThis problem is to design of a recommendation system based on each session. An session can be considered as a sequence of consecutive action taken by one customer based on a specific initiation reason. 'type' is category of action, '0' represents 'click', '1' is considered as 'add the item to cart', and '2' is as 'order the item'. 'aid' and 'ts' are the the corresponding page and timestamp. The goal of this problem is to predict potential pages related to each type of action, and use the prediction as recommendation.\n[Refer to competition page for evaluation metrics](https://www.kaggle.com/c/otto-recommender-system)","metadata":{}},{"cell_type":"markdown","source":"## Proposal\nWe plan to conduct the following steps to derive prediction: \n1. Create a word_list for each session, based on the combination of aid and type, the sequence of 'word' is based on timestamp\n2. Fit the word_list to word2vec, to calculat the simiarity among the 'word'\n3. Refine sessions to fit into word2vec model, in order to acheive higher performance","metadata":{}},{"cell_type":"code","source":"all = pl.concat([train, test])\nall.glimpse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:53:48.977688Z","iopub.execute_input":"2024-12-17T14:53:48.978119Z","iopub.status.idle":"2024-12-17T14:53:48.986778Z","shell.execute_reply.started":"2024-12-17T14:53:48.978080Z","shell.execute_reply":"2024-12-17T14:53:48.985427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# only select the sessions with type 2 action\nsessions_with_type_2 = all.filter(pl.col('type') == 2)['session'].unique()\nfiltered_df = all.filter(pl.col('session').is_in(sessions_with_type_2))\nfiltered_df_f = filtered_df.with_columns(pl.concat_str([\"aid\", \"type\"], separator=\"_\").alias(\"word\")).group_by(\"session\").agg(pl.col(\"word\").alias(\"word_list\"))\nsentences_filtered = filtered_df_f['word_list'].to_list()\nfiltered_df_f.glimpse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:53:08.586726Z","iopub.status.idle":"2024-12-17T14:53:08.587281Z","shell.execute_reply.started":"2024-12-17T14:53:08.587005Z","shell.execute_reply":"2024-12-17T14:53:08.587033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nw2vec_filter2 = Word2Vec(sentences=sentences_filtered, vector_size=32, min_count=1, workers=4)\n\nw2vec_filter2.save(\"word2vec_model_filter2.model\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:53:08.588949Z","iopub.status.idle":"2024-12-17T14:53:08.589497Z","shell.execute_reply.started":"2024-12-17T14:53:08.589223Z","shell.execute_reply":"2024-12-17T14:53:08.589251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#prepare trainig dataset \ntest_data = test['session'].unique()\ntest_data_df = test.filter(pl.col('session').is_in(test_data))\ntest_df_f = test_data_df.with_columns(pl.concat_str([\"aid\", \"type\"], separator=\"_\").alias(\"word\")).group_by(\"session\").agg(pl.col(\"word\").alias(\"word_list\"))\n\ntest_data_series = pl.Series(\"session\", test_data)\nordered_sessions = pl.DataFrame({\"session\": test_data_series})\ntest_df_f = ordered_sessions.join(test_df_f,on=\"session\",how=\"left\")\n\ntest_sentence = test_df_f['word_list'].to_list()\n\ntest_df_f.glimpse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:53:30.795537Z","iopub.execute_input":"2024-12-17T14:53:30.796575Z","iopub.status.idle":"2024-12-17T14:53:30.803510Z","shell.execute_reply.started":"2024-12-17T14:53:30.796519Z","shell.execute_reply":"2024-12-17T14:53:30.802371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nw2vec_filter2.wv.most_similar(test_df_f['word_list'][0].to_list(), topn = 20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:41:17.041304Z","iopub.execute_input":"2024-12-17T15:41:17.041799Z","iopub.status.idle":"2024-12-17T15:41:17.194403Z","shell.execute_reply.started":"2024-12-17T15:41:17.041724Z","shell.execute_reply":"2024-12-17T15:41:17.193130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sessions = test['session'].unique()\ntypes = ['clicks', 'carts', 'orders']\nsession_type_combinations = [f\"{session}_{type}\" for session in test_data for type in types]\n\noutput = pd.DataFrame(session_type_combinations)\nlen(test_sentence),output.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:53:08.595655Z","iopub.status.idle":"2024-12-17T14:53:08.596224Z","shell.execute_reply.started":"2024-12-17T14:53:08.595937Z","shell.execute_reply":"2024-12-17T14:53:08.595966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#we noticed that somes word shown in test data file may not be part of word2vec vocabulary, we will first clean up our test data by removing such cases\nclean_test_sentence = []\n\ndef clean_list(number_list, w2vec_model):\n    cleaned_list = []\n    for number in number_list:\n        if str(number) in w2vec_model.wv:\n            cleaned_list.append(str(number))\n    return cleaned_list\n\nfor sentence in test_sentence:\n     cleaned_list = clean_list(sentence,w2vec_filter2)\n     clean_test_sentence.append(cleaned_list)\n\nlen(clean_test_sentence)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:55:46.671775Z","iopub.execute_input":"2024-12-17T14:55:46.672209Z","iopub.status.idle":"2024-12-17T14:56:00.635689Z","shell.execute_reply.started":"2024-12-17T14:55:46.672172Z","shell.execute_reply":"2024-12-17T14:56:00.634333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"empty_elements = [elem for elem in clean_test_sentence if not elem]\nlen(empty_elements)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T14:56:18.751147Z","iopub.execute_input":"2024-12-17T14:56:18.751533Z","iopub.status.idle":"2024-12-17T14:56:18.811134Z","shell.execute_reply.started":"2024-12-17T14:56:18.751501Z","shell.execute_reply":"2024-12-17T14:56:18.810025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = test_df_f.with_columns(pl.Series('clean_word_list',clean_test_sentence))\ndf.glimpse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:45:51.977481Z","iopub.execute_input":"2024-12-17T15:45:51.977909Z","iopub.status.idle":"2024-12-17T15:45:51.985615Z","shell.execute_reply.started":"2024-12-17T15:45:51.977870Z","shell.execute_reply":"2024-12-17T15:45:51.984433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"w2vec_filter2.wv.most_similar(df['clean_word_list'][0].to_list(), topn = 20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:52:17.627586Z","iopub.execute_input":"2024-12-17T15:52:17.628005Z","iopub.status.idle":"2024-12-17T15:52:17.747865Z","shell.execute_reply.started":"2024-12-17T15:52:17.627968Z","shell.execute_reply":"2024-12-17T15:52:17.746759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = df[0:5000]\ntest_data_1 = test_data[0:5000]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:07:55.053681Z","iopub.execute_input":"2024-12-17T16:07:55.054089Z","iopub.status.idle":"2024-12-17T16:07:55.060178Z","shell.execute_reply.started":"2024-12-17T16:07:55.054055Z","shell.execute_reply":"2024-12-17T16:07:55.058387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"types = ['clicks', 'carts', 'orders']\nsession_type_combinations = [f\"{session}_{type}\" for session in test_data_1 for type in types]\n\noutput = pd.DataFrame(session_type_combinations)\nlen(test['clean_word_list']),output.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:07:56.568949Z","iopub.execute_input":"2024-12-17T16:07:56.569451Z","iopub.status.idle":"2024-12-17T16:07:56.584978Z","shell.execute_reply.started":"2024-12-17T16:07:56.569403Z","shell.execute_reply":"2024-12-17T16:07:56.583713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def remove_last_two_digits(input_string):\n    elements = input_string.split()\n    modified_elements = [element[:-2] for element in elements]\n    result = ' '.join(modified_elements) \n    return result\n\nlabels = []\nfor i in range(test.shape[0]):\n   if test['clean_word_list'][i].to_list():\n    similar_items = w2vec_filter2.wv.most_similar(test['clean_word_list'][i].to_list(), topn=20)\n    pred = ' '.join(str(item[0]) for item in similar_items)\n    output_string = remove_last_two_digits(pred)\n    #labels.extend(similar_items * 3)\n    labels.append(output_string)\n    labels.append(output_string)\n    labels.append(output_string)\n   else:# Handle the case when the sentence is empty\n        labels.append(\"911336\") #to refine later \n        labels.append(\"911336\")\n        labels.append(\"911336\")\n\noutput['labels'] = labels\noutput.tail(10)\n   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:08:06.793551Z","iopub.execute_input":"2024-12-17T16:08:06.794020Z","iopub.status.idle":"2024-12-17T16:15:43.115467Z","shell.execute_reply.started":"2024-12-17T16:08:06.793978Z","shell.execute_reply":"2024-12-17T16:15:43.112880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#refine labels for each category, based on action type\n#rank order output based on last digit before removing last two digit\n# for orders: keep top 5\n# for carts: keep top 10\n# for clicks: keep top 20\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = df[0:20]\ntest_data_1 = test_data[0:20]\n\ntypes = ['clicks', 'carts', 'orders']\nsession_type_combinations = [f\"{session}_{type}\" for session in test_data_1 for type in types]\n\noutput = pd.DataFrame(session_type_combinations)\nlen(test['clean_word_list']),output.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:35:25.633007Z","iopub.execute_input":"2024-12-17T17:35:25.633406Z","iopub.status.idle":"2024-12-17T17:35:25.642907Z","shell.execute_reply.started":"2024-12-17T17:35:25.633369Z","shell.execute_reply":"2024-12-17T17:35:25.641561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rank_words(word_list):\n  return sorted(word_list, key=lambda word: int(word[-1]), reverse=True)\n\nlabels = []\nraw = []\norigin = []\nfor i in range(test.shape[0]):\n   if test['clean_word_list'][i].to_list():\n    similar_items = w2vec_filter2.wv.most_similar(test['clean_word_list'][i].to_list(), topn=30)\n    pred = ' '.join(str(item[0]) for item in similar_items)\n    pred_1 = pred.split(' ')\n    rank_similar_items = rank_words(pred_1)\n    pred_clicks = rank_similar_items[:20]\n    pred_carts = rank_similar_items[:10]\n    pred_orders = rank_similar_items[:5]\n    output_string_clicks = [word[:-2] for word in pred_clicks]\n    output_string_carts = [word[:-2] for word in pred_carts]\n    output_string_orders = [word[:-2] for word in pred_orders]\n    #labels.extend(similar_items * 3)\n    labels.append(output_string_clicks)\n    labels.append(output_string_carts)\n    labels.append(output_string_orders)\n    raw.append(pred_clicks)\n    raw.append(pred_clicks)\n    raw.append(pred_clicks)\n    origin.append(similar_items)\n    origin.append(similar_items)\n    origin.append(similar_items)\n   else:# Handle the case when the sentence is empty\n        labels.append(\"911336\") #to refine later \n        labels.append(\"911336\")\n        labels.append(\"911336\")\n\noutput['labels'] = labels\noutput['raw'] = raw\noutput['origin'] = origin\noutput = output.rename(columns = {0:'session'})\noutput.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:35:27.403097Z","iopub.execute_input":"2024-12-17T17:35:27.403486Z","iopub.status.idle":"2024-12-17T17:35:29.458187Z","shell.execute_reply.started":"2024-12-17T17:35:27.403449Z","shell.execute_reply":"2024-12-17T17:35:29.456899Z"}},"outputs":[],"execution_count":null}]}