{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# IMPORTING LIBRARIES","metadata":{"id":"PsSpcV7sx4t0"}},{"cell_type":"code","source":"# importing necessary Python libraries\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport plotly.express as px\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\nimport os\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom datetime import datetime\nimport matplotlib.pyplot as plt\nfrom itertools import chain\nimport plotly.graph_objs as go \n#import plotly.figure_factory as ff\n\n# avoid displaying warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n#import machine learning related libraries\nfrom sklearn.svm import SVC\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import KFold, cross_val_score, train_test_split, GridSearchCV, cross_validate\nfrom sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score, confusion_matrix\nfrom sklearn.cluster import KMeans\nimport xgboost as xgb\nimport time \n","metadata":{"id":"xU090_CgjUxk","execution":{"iopub.status.busy":"2022-12-01T06:38:43.304056Z","iopub.execute_input":"2022-12-01T06:38:43.304370Z","iopub.status.idle":"2022-12-01T06:38:43.314421Z","shell.execute_reply.started":"2022-12-01T06:38:43.304337Z","shell.execute_reply":"2022-12-01T06:38:43.313598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IMPORTING THE DATAFRAMES ","metadata":{"id":"cUCkgG-JtGnd"}},{"cell_type":"code","source":"\n# import customer.csv , transactions_train.csv, articles.csv, sample_submission.csv from https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/data\n\ndf_customers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\ndf_transactions = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ndf_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_sample_submission = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')","metadata":{"id":"oBW7TwdpjZ_j","execution":{"iopub.status.busy":"2022-12-01T06:38:43.316432Z","iopub.execute_input":"2022-12-01T06:38:43.316801Z","iopub.status.idle":"2022-12-01T06:40:05.940007Z","shell.execute_reply.started":"2022-12-01T06:38:43.316764Z","shell.execute_reply":"2022-12-01T06:40:05.938296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cleaning the customers table\n#### Replacing age NaN values with 0\n","metadata":{"id":"sgka1GohvUlR"}},{"cell_type":"code","source":"df_customers = df_customers.dropna(subset=['age']) \ndf_customers.isna().sum()","metadata":{"id":"KfQuZVoBvjCE","outputId":"266f6e1e-08ef-4fcc-e132-d96a5aaea338","execution":{"iopub.status.busy":"2022-12-01T06:40:05.946046Z","iopub.execute_input":"2022-12-01T06:40:05.947192Z","iopub.status.idle":"2022-12-01T06:40:06.518020Z","shell.execute_reply.started":"2022-12-01T06:40:05.947139Z","shell.execute_reply":"2022-12-01T06:40:06.516551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_12 = df_transactions.customer_id.value_counts()\ncustomer_best = customers_12.index[customers_12 > 12] ","metadata":{"execution":{"iopub.status.busy":"2022-12-01T06:40:06.519450Z","iopub.execute_input":"2022-12-01T06:40:06.520744Z","iopub.status.idle":"2022-12-01T06:40:15.709883Z","shell.execute_reply.started":"2022-12-01T06:40:06.520688Z","shell.execute_reply":"2022-12-01T06:40:15.708480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Taking 10000 rows for sample data undertaking\ncustomer_sample = df_customers[df_customers['customer_id'].isin(customer_best)].sample(n=10000, frac=None, replace=False, weights=None, random_state=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-01T06:40:15.714000Z","iopub.execute_input":"2022-12-01T06:40:15.714314Z","iopub.status.idle":"2022-12-01T06:40:16.983471Z","shell.execute_reply.started":"2022-12-01T06:40:15.714277Z","shell.execute_reply":"2022-12-01T06:40:16.982089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum().sort_values(ascending = False)\n    percent = (data.isnull().sum()/data.isnull().count()*100).sort_values(ascending = False)\n    return pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n\n","metadata":{"id":"fIgFvOhCwZME","execution":{"iopub.status.busy":"2022-12-01T06:40:16.986089Z","iopub.execute_input":"2022-12-01T06:40:16.986466Z","iopub.status.idle":"2022-12-01T06:40:16.994802Z","shell.execute_reply.started":"2022-12-01T06:40:16.986417Z","shell.execute_reply":"2022-12-01T06:40:16.993261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(df_articles).head(7).style.set_properties(**{'background-color': 'rgba(245, 181, 152,.5)'})","metadata":{"id":"T1PRw5J8wcrf","outputId":"d1ffa4b9-5657-48ff-9b3b-c9d6d2d3913c","execution":{"iopub.status.busy":"2022-12-01T06:40:16.996144Z","iopub.execute_input":"2022-12-01T06:40:16.996399Z","iopub.status.idle":"2022-12-01T06:40:17.247781Z","shell.execute_reply.started":"2022-12-01T06:40:16.996369Z","shell.execute_reply":"2022-12-01T06:40:17.245943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FEATURE SELECTION & FEATURE ENGINEERING","metadata":{"id":"dNtX-2mc0Edp"}},{"cell_type":"markdown","source":"#### Data splitting is among the most critical steps before preprocessing. It is possible to cause data leakage by dividing the data before processing it, or to overestimate the model evaluation when splitting the data. But Splitting data prior to processing has the huge advantage of ensuring consistency in model performance because unseen data are processed in the same manner as test data.","metadata":{"id":"_m5hRi760X1r"}},{"cell_type":"code","source":"#Split customer sample data (10000 rows) in test and train dataframe\ncustomer_train, customer_test, = train_test_split(customer_sample, test_size=0.3)","metadata":{"id":"opJ25Er30krX","execution":{"iopub.status.busy":"2022-12-01T06:40:17.249750Z","iopub.execute_input":"2022-12-01T06:40:17.250139Z","iopub.status.idle":"2022-12-01T06:40:17.263639Z","shell.execute_reply.started":"2022-12-01T06:40:17.250096Z","shell.execute_reply":"2022-12-01T06:40:17.262576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customer_train.head()\ncustomer_test.shape \ncustomer_train.shape ","metadata":{"id":"eoa7nKnC0t7T","outputId":"22f26edb-5539-4bb0-c1c6-1eab61347db7","execution":{"iopub.status.busy":"2022-12-01T06:40:17.265570Z","iopub.execute_input":"2022-12-01T06:40:17.266045Z","iopub.status.idle":"2022-12-01T06:40:17.273979Z","shell.execute_reply.started":"2022-12-01T06:40:17.266006Z","shell.execute_reply":"2022-12-01T06:40:17.272985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data frame df_articles has been cleaned up by removing extra columns. We need to drop a couple of columns from the df_articles data frame since it is extremely large, and we won't be evaluating the machine learning model with those columns.","metadata":{"id":"GjBsefM9yTsq"}},{"cell_type":"code","source":"# remove unwanted columns from articles df and place it into a new df\nclean_df_articles = df_articles.drop(columns=['prod_name', 'product_type_name', 'graphical_appearance_name',\n                                       'colour_group_name', 'perceived_colour_value_name', 'perceived_colour_master_name', 'department_name', \n                                'perceived_colour_value_id', 'perceived_colour_master_id',\n                                       'index_name', 'index_group_name', 'section_name', 'garment_group_name'])","metadata":{"id":"NpoYcR1E45e9","execution":{"iopub.status.busy":"2022-12-01T06:40:17.275771Z","iopub.execute_input":"2022-12-01T06:40:17.276057Z","iopub.status.idle":"2022-12-01T06:40:17.291928Z","shell.execute_reply.started":"2022-12-01T06:40:17.276021Z","shell.execute_reply":"2022-12-01T06:40:17.291091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_df_articles.info()","metadata":{"id":"pH4REVUuygFl","outputId":"5e55a671-68ef-4300-8b9e-9c9c409b6f45","execution":{"iopub.status.busy":"2022-12-01T06:40:17.293037Z","iopub.execute_input":"2022-12-01T06:40:17.293360Z","iopub.status.idle":"2022-12-01T06:40:17.337207Z","shell.execute_reply.started":"2022-12-01T06:40:17.293325Z","shell.execute_reply":"2022-12-01T06:40:17.336457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### From the above cell, we can see that unwated columns have been removed. Additionally, we convert the data types for the article column to categorical to reduce processing time.","metadata":{"id":"eMekiWCyyxaC"}},{"cell_type":"code","source":"clean_df_articles['product_code'] = clean_df_articles.product_code.astype('category')\nclean_df_articles['product_type_no'] = clean_df_articles.product_type_no.astype('category')\nclean_df_articles['product_group_name'] = clean_df_articles.product_group_name.astype('category')\nclean_df_articles['graphical_appearance_no'] = clean_df_articles.graphical_appearance_no.astype('category')\nclean_df_articles['colour_group_code'] = clean_df_articles.colour_group_code.astype('category')\nclean_df_articles['department_no'] = clean_df_articles.department_no.astype('category')\nclean_df_articles['index_code'] = clean_df_articles.product_type_no.astype('category')\nclean_df_articles['index_group_no'] = clean_df_articles.index_group_no.astype('category')\nclean_df_articles['section_no'] = clean_df_articles.section_no.astype('category')\nclean_df_articles['garment_group_no'] = clean_df_articles.garment_group_no.astype('category')\nclean_df_articles['detail_desc'] = clean_df_articles.detail_desc.astype('category')","metadata":{"id":"3Sa5vqyU5lAK","execution":{"iopub.status.busy":"2022-12-01T06:40:17.338946Z","iopub.execute_input":"2022-12-01T06:40:17.339204Z","iopub.status.idle":"2022-12-01T06:40:17.528686Z","shell.execute_reply.started":"2022-12-01T06:40:17.339173Z","shell.execute_reply":"2022-12-01T06:40:17.527188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_df_articles.info()","metadata":{"id":"ZE1PL_4t5hWw","outputId":"d381abcb-6cba-44ae-d735-eea8c3c21c2a","execution":{"iopub.status.busy":"2022-12-01T06:40:17.530544Z","iopub.execute_input":"2022-12-01T06:40:17.530919Z","iopub.status.idle":"2022-12-01T06:40:17.605100Z","shell.execute_reply.started":"2022-12-01T06:40:17.530880Z","shell.execute_reply":"2022-12-01T06:40:17.603809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge df_article and df_transastion table into a new df called article_transactions","metadata":{"id":"Lo_F01E31qAj"}},{"cell_type":"code","source":"article_transactions = df_transactions.merge(clean_df_articles, on='article_id', how='left')","metadata":{"id":"HHKReDcM6J2B","execution":{"iopub.status.busy":"2022-12-01T06:40:17.609365Z","iopub.execute_input":"2022-12-01T06:40:17.609737Z","iopub.status.idle":"2022-12-01T06:40:27.457664Z","shell.execute_reply.started":"2022-12-01T06:40:17.609697Z","shell.execute_reply":"2022-12-01T06:40:27.456191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_transactions.info()","metadata":{"id":"SrweSl4J9vp6","outputId":"75706e23-1132-4cda-d134-234ae3114b4b","execution":{"iopub.status.busy":"2022-12-01T06:40:27.459458Z","iopub.execute_input":"2022-12-01T06:40:27.459745Z","iopub.status.idle":"2022-12-01T06:40:27.474147Z","shell.execute_reply.started":"2022-12-01T06:40:27.459712Z","shell.execute_reply":"2022-12-01T06:40:27.472720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_transactions.head(2)","metadata":{"id":"uweln1jG1j3o","outputId":"19784fd3-2270-4efb-a9f4-c5601acca9b7","execution":{"iopub.status.busy":"2022-12-01T06:40:27.475688Z","iopub.execute_input":"2022-12-01T06:40:27.476401Z","iopub.status.idle":"2022-12-01T06:40:27.506246Z","shell.execute_reply.started":"2022-12-01T06:40:27.476339Z","shell.execute_reply":"2022-12-01T06:40:27.504610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"YuoFvHeT1qAi"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge customer training and customer test dataframe which we created earlier with article_transactions df to create a proper Train & Test Df which we will use for our testing and prediction.","metadata":{"id":"Wn9QIqQz1qbi"}},{"cell_type":"code","source":"# Merge customer_train,test with article_transactions to get train & test dataframe\ntrain = customer_train.merge(article_transactions, on='customer_id', how='inner') \ntest = customer_test.merge(article_transactions, on='customer_id', how='inner')","metadata":{"id":"wsZueHZC95uI","execution":{"iopub.status.busy":"2022-12-01T06:40:27.508073Z","iopub.execute_input":"2022-12-01T06:40:27.508856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()\ntest.info()","metadata":{"id":"ke_JxI0e95xX","outputId":"8fbad5c6-4daf-4d44-b3f1-78172b89d1a6","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(-2)","metadata":{"id":"vb0EiIJj9yaa","outputId":"09441461-d466-4436-bd4e-41857f291a34","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert Column t_dat in both train & test df to proper datetime format","metadata":{"id":"9xBbHo0n2KF5"}},{"cell_type":"code","source":"#Convert Column t_dat in both train & test df to proper datetime format\ntrain.t_dat = pd.to_datetime(train.t_dat)\ntest.t_dat = pd.to_datetime(test.t_dat)","metadata":{"id":"61ujlOQr_5WN","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Sort the customer_id values from test & train df\ntrain = train.sort_values(['customer_id', 't_dat'])\ntest = test.sort_values(['customer_id', 't_dat'])","metadata":{"id":"Tz7ugySj_5Y5","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get last 12 item buys of all the customers","metadata":{"id":"OZQwOt2a2lLj"}},{"cell_type":"code","source":"train_group = train.groupby('customer_id', observed=True).tail(12).index #List of the 12 last index\ntest_group = test.groupby('customer_id', observed=True).tail(12).index","metadata":{"id":"ru9wom8x_5by","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#replaces everything up to groupby in next cell\ntrain.loc[train_group,:] \ntest.loc[test_group,:]","metadata":{"id":"kOiiYyXN_5ed","outputId":"c291fef9-86f7-4d25-d220-401129a5689e","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We drop unwanted columns from train and test to get dataframes with customers and their last 12 purchases only.\nThis will give us a y_truth value which states real purchases by customers.","metadata":{"id":"RlvN1sjz6PIV"}},{"cell_type":"code","source":"#train.drop(index=train_group, inplace=True)\n#test.drop(index=test_group, inplace=True)","metadata":{"id":"6-gz-O_j5bOE","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train.loc[train_group,:].groupby('customer_id', observed=True)['article_id'].apply(lambda x: x.tolist())\ny_test =  test.loc[test_group,:].groupby('customer_id', observed=True)['article_id'].apply(lambda x: x.tolist())","metadata":{"id":"_Uq1kjDY_5hD","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.head(100)\ny_test.head(100)","metadata":{"id":"NGCFyxWHihs5","outputId":"edbfd2b8-ae07-4012-a1e3-68a0f3dab4a5","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create data frame for y_train & y_test to get value for just 1 customer\nuno_y_train = y_train.apply(lambda x: x[1])\nuno_y_test = y_test.apply(lambda x: x[1])","metadata":{"id":"6dtdbJfBOC3g","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uno_y_train.head()","metadata":{"id":"JcQQnxn1PlTl","outputId":"4cfef562-2164-481b-ad6b-864766082d27","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uno_y_test.head()","metadata":{"id":"3kTG8OFNNrjF","outputId":"eb078f8b-2551-44a7-c647-d4d3b5a0eb28","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to get X_train & x_test data frames. \nThis function will helpfull to finding the features values which will be used for finding accuracy.\n","metadata":{"id":"KLxswkt-Ayp2"}},{"cell_type":"code","source":"def function_features (customers):\n        #here we will fetch all the required columns which we will use as features\n        features_rows = {'FN' : customers['FN'].iloc[0],\n                    'Active' : customers['Active'].iloc[0],\n                    'club_member_status' : customers['club_member_status'].iloc[0],\n                    'fashion_news_frequency' :customers['fashion_news_frequency'].iloc[0],\n                    'age'  : customers['age'].iloc[0],\n                    'postal_code' : customers['postal_code'].iloc[0]} \n        features_rows['bought_items'] = customers.shape[0] #\n        return pd.Series(features_rows) \n        \n","metadata":{"id":"vkP1IYNMAgjz","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Getting x_train & x_test","metadata":{"id":"CzbgCbcgCiF7"}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\ntqdm.pandas()\n\nx_train = train.groupby('customer_id', observed=True).progress_apply(function_features) #Apply feat_gen to entire Full_train\n\nx_test  = test.groupby('customer_id', observed=True).progress_apply(function_features) ","metadata":{"id":"ZigTwkhiAgrq","outputId":"ab7b345a-5657-45b9-f621-3cd8719e3329","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.head()","metadata":{"id":"J9FgiFLdAguC","outputId":"815e5ccc-fe66-4034-c773-201d91429b58","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.head()","metadata":{"id":"4xGjwiwGAgwh","outputId":"f9e8bc31-a4db-4d59-c75b-c4299629359d","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now we get the prediction column (y_prediction) for training data\n","metadata":{"id":"PCmDRa5vDWeL"}},{"cell_type":"code","source":"train_prediction = train.groupby([\"customer_id\"])[\"article_id\"].agg(lambda x: str(x.values[0:12])[1:-1]).reset_index()","metadata":{"id":"NUD6b0WBAgy1","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets define a function to get prediction values by splitting all items values.","metadata":{"id":"-Czn_E5hDwsH"}},{"cell_type":"code","source":"def articles_padding(x):\n    if x:\n        xl = x.split()\n        x = []\n        for xi in xl:\n            x.append(\"0\"+xi)\n        dimm_x = len(x)\n        if dimm_x < 12:\n            x.extend(art_list[:12-dimm_x])\n        return(\" \".join(x))","metadata":{"id":"k_3PpZ97Ag1L","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_prediction[\"article_id\"] = train_prediction[\"article_id\"].apply(lambda x: articles_padding(x))","metadata":{"id":"hvlP34zbAg3j","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Temporary predictied articles ( we will get proper prediction later.)","metadata":{"id":"MVrAGfFNEj_R"}},{"cell_type":"code","source":"train_prediction.head()","metadata":{"id":"LbTEzIJkAg6m","outputId":"b8e1fa62-e1ac-4cca-9ff4-9ecfe91f4356","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now we get the prediction column (y_prediction) for test data\n","metadata":{"id":"C9mpJE0YFDGS"}},{"cell_type":"code","source":"test_prediction = test.groupby([\"customer_id\"])[\"article_id\"].agg(lambda x: str(x.values[0:12])[1:-1]).reset_index()","metadata":{"id":"iA0oZjd3E62y","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def articles_padding_test(x):\n    if x:\n        xl = x.split()\n        x = []\n        for xi in xl:\n            x.append(\"0\"+xi)\n        dimm_x = len(x)\n        if dimm_x < 12:\n            x.extend(art_list[:12-dimm_x])\n        return(\" \".join(x))","metadata":{"id":"ux3Xr-VrE7KJ","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prediction[\"article_id\"] = test_prediction[\"article_id\"].apply(lambda x: articles_padding_test(x))","metadata":{"id":"MwhyX3TrE7OY","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prediction.head()","metadata":{"id":"HGI6VAQ4FH7b","outputId":"363a7a42-055e-418e-8ea4-156efc3bcba3","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Machine Learning Algorithm to Find accuracy and Predcition of the training and test models.","metadata":{"id":"PSJTV7_gErCu"}},{"cell_type":"markdown","source":"Before we proceed with ml part , we will need to convert our values in x_train & x_test to one hot encoding because currently we have all the data types as categorical which will be needed to converted to int or float to proceed with predcting accuracy","metadata":{"id":"wp5V-dc7FX-4"}},{"cell_type":"code","source":"# remove na's in x_train\nx_train.fillna(0)","metadata":{"id":"RqRm5Jxf8DGS","outputId":"b60f59be-ae59-42d2-82f0-af01daaeb921","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove na's in x_test\nx_test.fillna(0)","metadata":{"id":"D00Pv5x08DI3","outputId":"6569204b-c24f-4b95-c5ae-91f6411badd1","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LETS CONVERT 'FN', 'Active', 'club_member_status', 'fashion_news_frequency' columns (these are categorical columns) to binary using one hot encoding in x_train and x_test","metadata":{"id":"fWevH12YGMYd"}},{"cell_type":"code","source":"# applying the one hot encoding  to our X train and  X test dataframe  for 'FN', 'Active', 'club_member_status', 'fashion_news_frequency' columns\nx_train_encoded = pd.get_dummies(x_train, columns=['FN', 'Active', 'club_member_status', 'fashion_news_frequency'])\nx_test_encoded = pd.get_dummies(x_test, columns=['FN', 'Active', 'club_member_status', 'fashion_news_frequency'])","metadata":{"id":"8eID61M38DLS","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lets drop unwated column (postal_code, as it is not needed)\nx_train_encoded.drop(['postal_code'], axis=1, inplace=True)\nx_test_encoded.drop(['postal_code'], axis=1, inplace=True)","metadata":{"id":"2uvJMw-GGLyT","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nx_train_encoded.head()\n","metadata":{"id":"ef7vySuTGL4D","outputId":"7b7c45d6-b68b-497b-c9b6-e1a2ebdf79fa","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test_encoded.head()","metadata":{"id":"HIEVtV8aGL9F","outputId":"c8c40bd9-db68-4618-b9c0-bcad944c970e","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that all categorical columns are now converted to binary 1's & 0's to make our job easier to get accuracy","metadata":{"id":"rYotvzmCHKW5"}},{"cell_type":"markdown","source":"## Now lets find accuracy using customers data which we did to get a value","metadata":{"id":"3WuYo3w_HXqb"}},{"cell_type":"markdown","source":"# i. DECISION TREE","metadata":{"id":"JuZJvQOxW3lw"}},{"cell_type":"code","source":"\nfrom sklearn.tree import DecisionTreeClassifier # Import Decision Tree Classifier\nfrom sklearn.model_selection import train_test_split # Import train_test_split function\nfrom sklearn import metrics #Import scikit-learn metrics module for accuracy calculation","metadata":{"id":"PfQJzt2kIzQm","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Decision tree classifer object\nclf = DecisionTreeClassifier()\n\n# Train Decision Tree Classifer\nclf = clf.fit(x_train_encoded,uno_y_train)\n\n#Predict the response for test dataset\ny_pred = clf.predict(x_train_encoded)","metadata":{"id":"2Qd7TXo8IzTa","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_train)","metadata":{"id":"h1i2JpCGM1sM","outputId":"3760e7ac-1a97-4836-dd15-b33c58ee5571","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred)","metadata":{"id":"uK1TX9AOSvYW","outputId":"9b8928d3-930f-4cfe-95b5-9132bd2f7518","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ** WE GET A ACCURACY OF 60 % using Decison Tree WHICH TELLS US THAT OUR PREDICTION WAS CORRECT**\n\n---\n\n","metadata":{"id":"rtYWyfmDXG1z"}},{"cell_type":"code","source":"print(\"Accuracy:\",metrics.accuracy_score(uno_y_train, y_pred))","metadata":{"id":"cLYu4AkiSxnF","outputId":"37ff74ad-4704-4e23-fe59-05e42a72eef8","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ii. KNN Algorithm","metadata":{"id":"r_0AqM1YXYQm"}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\n\nX = x_train_encoded\ny = uno_y_train\n\nKnnc = KNeighborsClassifier(n_neighbors=5)\nKnnc.fit(X,y)\ny_pred = Knnc.predict(x_train_encoded)\n\nprint(accuracy_score(uno_y_train, y_pred)) \n","metadata":{"id":"1_F1o--ELM9l","outputId":"5e2fcf5c-1d32-41b9-f1a9-a92178731eaa","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets Get single predictions  for all customer using following function","metadata":{"id":"9qF2HFw6ZWeW"}},{"cell_type":"code","source":"#allx = pd.concat([x_train_encoded, x_test_encoded]) \n#average_customer = allx.mean(axis=0).to_frame().T\n#missing_new = df_customers['customer_id'][~df_customers['customer_id'].isin(allx.index)]\n#submissions = clf.predict(allx)\n#customer_ppred = clf.predict(average_customer)","metadata":{"id":"yDQOUJuiIOYZ","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submissions = pd.DataFrame({'customer_id': allx.index, 'prediction' : submissions})\n#submissions = pd.concat([submissions, pd.DataFrame({'customer_id': missing_new, 'prediction': np.repeat(customer_ppred, len(missing_new))})])","metadata":{"id":"0As2trOHaCin","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submissions.sort_index(inplace=True)","metadata":{"id":"nhTjIF22djWB","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets Get 12 prediction for all customers using following function","metadata":{"id":"PxBeWaP-eb_N"}},{"cell_type":"code","source":"pred_df = df_transactions.groupby([\"customer_id\"])[\"article_id\"].agg(lambda x: str(x.values[0:12])[1:-1]).reset_index()","metadata":{"id":"B78rk0Tbef38","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find customers with purchases on last days\nlast_date = df_transactions.t_dat.max()\nprint(df_transactions.loc[df_transactions.t_dat==last_date].shape)\n\n\n# find most frequent items\nmost_frequent_articles = list(df_transactions.loc[df_transactions.t_dat==last_date].article_id.value_counts()[0:12].index)\nart_list = []\nfor art in most_frequent_articles:\n    art = \"0\"+str(art)\n    art_list.append(art)\nart_str = \" \".join(art_list)\nprint(\"Frequent articles bought recently:\", art_str, end=\"\\n\")","metadata":{"id":"5LlYwcwNfSoT","outputId":"b4cd9d41-0da5-40d8-b8ec-9dadd047c51e","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def padding_articles_prediction(x):\n    if x:\n        xl = x.split()\n        x = []\n        for xi in xl:\n            x.append(\"0\"+xi)\n        dimm_x = len(x)\n        if dimm_x < 12:\n            x.extend(art_list[:12-dimm_x])\n        return(\" \".join(x))","metadata":{"id":"SP4edQq_ejk3","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df[\"article_id\"] = pred_df[\"article_id\"].apply(lambda x: padding_articles_prediction(x))\n","metadata":{"id":"Vf2zcI6YfAYI","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace sample submission files with our predicted values\ndf_submission = pred_df.merge(df_sample_submission[[\"customer_id\"]], how=\"right\")\ndf_submission.columns = [\"customer_id\", \"prediction\"]\ndf_submission.head().style.set_properties(**{'background-color': 'rgba(184,230,194,.5)'})","metadata":{"id":"0k2M7xCMfGQk","outputId":"066ec9d9-b2d4-4c88-f424-cbae946001bf","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv',index=False)","metadata":{"id":"JlvpZ1l0f2mh","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\n\ndef create_download_link(title = \"Download CSV file\", filename = \"data.csv\"):  \n    html = '<a href= >{title}</a >'\n    html = html.format(title=title,filename=filename)\n    return HTML(html)\n\n# create a link to download the dataframe which was saved with .to_csv method\ncreate_download_link(filename='submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\n\ndef create_download_link(title = \"Download CSV file\", filename = \"data.csv\"):  \n    html = '<a href= >{title}</a >'\n    html = html.format(title=title,filename=filename)\n    return HTML(html)\n\n# create a link to download the dataframe which was saved with .to_csv method\ncreate_download_link(filename='submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}