{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# IMPORTING LIBRARIES","metadata":{"id":"PsSpcV7sx4t0"}},{"cell_type":"code","source":"# importing necessary Python libraries\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport plotly.express as px\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\nimport os\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom datetime import datetime\nimport matplotlib.pyplot as plt\nfrom itertools import chain\nimport plotly.graph_objs as go \n#import plotly.figure_factory as ff\n\n# avoid displaying warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n#import machine learning related libraries\nfrom sklearn.svm import SVC\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import KFold, cross_val_score, train_test_split, GridSearchCV, cross_validate\nfrom sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score, confusion_matrix\nfrom sklearn.cluster import KMeans\nimport xgboost as xgb\nimport time \n","metadata":{"id":"xU090_CgjUxk","execution":{"iopub.status.busy":"2022-05-05T04:45:18.890068Z","iopub.execute_input":"2022-05-05T04:45:18.890399Z","iopub.status.idle":"2022-05-05T04:45:18.902599Z","shell.execute_reply.started":"2022-05-05T04:45:18.890361Z","shell.execute_reply":"2022-05-05T04:45:18.901588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IMPORTING THE DATAFRAMES ","metadata":{"id":"cUCkgG-JtGnd"}},{"cell_type":"code","source":"\n# import customer.csv , transactions_train.csv, articles.csv, sample_submission.csv from https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/data\n\ndf_customers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\ndf_transactions = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ndf_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_sample_submission = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')","metadata":{"id":"oBW7TwdpjZ_j","execution":{"iopub.status.busy":"2022-05-05T04:45:18.904526Z","iopub.execute_input":"2022-05-05T04:45:18.904814Z","iopub.status.idle":"2022-05-05T04:46:30.492931Z","shell.execute_reply.started":"2022-05-05T04:45:18.904776Z","shell.execute_reply":"2022-05-05T04:46:30.491569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{"id":"tB7c8eCNtL_8"}},{"cell_type":"markdown","source":"## $\\color{red}{\\text{1. Customers table  Exploration}}$ ","metadata":{"id":"4BXF8b41taqG"}},{"cell_type":"code","source":"\ndf_customers.info()\nprint('-------------------------------')\nprint('-------------------------------')\ndf_customers.shape[0] - df_customers['customer_id'].nunique()\nprint(\"Duplicate values:\",df_customers.shape[0] - df_customers['customer_id'].nunique())","metadata":{"id":"G2e1AprytfTm","outputId":"93297b2b-7fea-46c6-9f98-2d3bacc576bb","execution":{"iopub.status.busy":"2022-05-05T04:46:30.495135Z","iopub.execute_input":"2022-05-05T04:46:30.495446Z","iopub.status.idle":"2022-05-05T04:46:32.531891Z","shell.execute_reply.started":"2022-05-05T04:46:30.495407Z","shell.execute_reply":"2022-05-05T04:46:32.530466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Customers data description:\n\ncustomer_id : A unique identifier of every customer<br>\nFN : 1 or missed<br>\nActive : 1 or missed<br>\nclub_member_status : Status in club<br>\nfashion_news_frequency : How often H&M may send news to customer<br>\nage : The current age<br>\npostal_code : Postal code of customer<br>","metadata":{"id":"xdyKE5-XtmPd"}},{"cell_type":"markdown","source":"#### We can see there are some null values in customers columns: 'FN','age','Active','club_member_status','fashion_news_frequency'\n#### And column 'fashion_news_frequency' has 2 'None' values instead of 'NONE'\n#### There are no duplicate values in customers\n\n","metadata":{"id":"h8mW8ZjhtrSq"}},{"cell_type":"markdown","source":"## Lets Explore some insights about customers \n\n\n","metadata":{"id":"IBhs9AzcuK6s"}},{"cell_type":"markdown","source":"#### i. Number of Customers per each Age\nThe most common age is about 21-23\n\n","metadata":{"id":"cJHASM9It3mg"}},{"cell_type":"code","source":"def pie_chart(df, col_values, labels, ax, color, title):\n    n_classes = len(df)\n    explode = (0.5,) * n_classes # explode for 0.1 each slice\n    ax.pie(df[col_values],\n           colors=color, \n           explode=explode,\n           labels=df[labels],\n           shadow=True)\n    ax.set_title(title, fontsize=16)\n    \ndef bar_plot(df, col_x, col_y, ax, color, title):\n    ax.bar(x=df[col_x],\n           height=df[col_y],\n           color=color)\n    ax.set_title(title, fontsize=16) \n    plt.xticks(rotation=90)\n    \ntemp = df_customers.groupby([\"age\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Age': temp.index,'Customers': temp.values})\ndf = df.sort_values(['Age'], ascending=False)\n\n\nfig, axes = plt.subplots(nrows=1, ncols=1, figsize=(10,6))\ncolor = plt.cm.cool(np.linspace(0, 1, len(df)))\n\nbar_plot(df,\n         'Age',\n         'Customers',\n         axes, \n         color, \n         \"Number of Customers by age\")\n","metadata":{"id":"AATN0JKNt7gm","outputId":"0cf9db46-86ea-4844-da57-52e2716a4d64","execution":{"iopub.status.busy":"2022-05-05T04:46:32.534504Z","iopub.execute_input":"2022-05-05T04:46:32.534766Z","iopub.status.idle":"2022-05-05T04:46:33.122854Z","shell.execute_reply.started":"2022-05-05T04:46:32.534731Z","shell.execute_reply":"2022-05-05T04:46:33.121676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ii. Customer Status in H&M club. \nAlmost every customer has an active club status, some of them begin to activate it (pre-create). A tiny part of customers abandoned the club.","metadata":{"id":"cr_yAWk_vM_h"}},{"cell_type":"code","source":"sns.set_style(\"darkgrid\")\nf, ax = plt.subplots(figsize=(10,5))\nax = sns.histplot(data=df_customers, x='club_member_status', color='blue')\nax.set_xlabel('Distribution of club member status')\nplt.show()","metadata":{"id":"f95pwypCvOMk","outputId":"d5e04e38-66e5-4002-ceb1-0f9cc454b6ba","execution":{"iopub.status.busy":"2022-05-05T04:46:33.124626Z","iopub.execute_input":"2022-05-05T04:46:33.124870Z","iopub.status.idle":"2022-05-05T04:46:35.234953Z","shell.execute_reply.started":"2022-05-05T04:46:33.124840Z","shell.execute_reply":"2022-05-05T04:46:35.233718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cleaning the customers table\n#### Replacing age NaN values with 0\n","metadata":{"id":"sgka1GohvUlR"}},{"cell_type":"code","source":"df_customers = df_customers.dropna(subset=['age']) \ndf_customers.isna().sum()","metadata":{"id":"KfQuZVoBvjCE","outputId":"266f6e1e-08ef-4fcc-e132-d96a5aaea338","execution":{"iopub.status.busy":"2022-05-05T04:46:35.236306Z","iopub.execute_input":"2022-05-05T04:46:35.236606Z","iopub.status.idle":"2022-05-05T04:46:35.953287Z","shell.execute_reply.started":"2022-05-05T04:46:35.236565Z","shell.execute_reply":"2022-05-05T04:46:35.952116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Counting the 12 best customers","metadata":{"id":"T6kHWcjvvlwX"}},{"cell_type":"code","source":"customers_12 = df_transactions.customer_id.value_counts()","metadata":{"id":"eKA2oXPRvsfK","execution":{"iopub.status.busy":"2022-05-05T04:46:35.954777Z","iopub.execute_input":"2022-05-05T04:46:35.955163Z","iopub.status.idle":"2022-05-05T04:46:46.572919Z","shell.execute_reply.started":"2022-05-05T04:46:35.955118Z","shell.execute_reply":"2022-05-05T04:46:46.571476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"cutomers who bought more than 12 items","metadata":{"id":"rvq4h5_hwAko"}},{"cell_type":"code","source":"customers_12 > 12 \ncustomers_12.index[customers_12 > 12] #","metadata":{"id":"nV4FkTVGv49w","outputId":"4cca4270-528a-4150-dbf2-f1186ce38e12","execution":{"iopub.status.busy":"2022-05-05T04:46:46.574904Z","iopub.execute_input":"2022-05-05T04:46:46.575251Z","iopub.status.idle":"2022-05-05T04:46:46.739568Z","shell.execute_reply.started":"2022-05-05T04:46:46.575207Z","shell.execute_reply":"2022-05-05T04:46:46.738365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df with list of customer with more than 12 purchases\ncustomer_best = customers_12.index[customers_12 > 12] \ncustomer_best","metadata":{"id":"pUJaJ4DEwmdk","outputId":"9807934f-938f-4ddb-fd7a-301a5d5f290b","execution":{"iopub.status.busy":"2022-05-05T04:46:46.742834Z","iopub.execute_input":"2022-05-05T04:46:46.743207Z","iopub.status.idle":"2022-05-05T04:46:46.902333Z","shell.execute_reply.started":"2022-05-05T04:46:46.743168Z","shell.execute_reply":"2022-05-05T04:46:46.901205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Taking 10000 rows for sample data undertaking\ncustomer_sample = df_customers[df_customers['customer_id'].isin(customer_best)].sample(n=10000, frac=None, replace=False, weights=None, random_state=1)\n","metadata":{"id":"9KuLkzK7wxQH","execution":{"iopub.status.busy":"2022-05-05T04:46:46.903731Z","iopub.execute_input":"2022-05-05T04:46:46.904362Z","iopub.status.idle":"2022-05-05T04:46:48.054744Z","shell.execute_reply.started":"2022-05-05T04:46:46.904290Z","shell.execute_reply":"2022-05-05T04:46:48.052958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## $\\color{red}{\\text{2. Articles table  Exploration }}$ ","metadata":{"id":"k7wjKjRdwG-x"}},{"cell_type":"code","source":"df_articles.info()","metadata":{"id":"_xYLPuRdwGZ3","outputId":"5bf7b27a-62b5-4189-d1d1-547950655c7c","execution":{"iopub.status.busy":"2022-05-05T04:46:48.056677Z","iopub.execute_input":"2022-05-05T04:46:48.057246Z","iopub.status.idle":"2022-05-05T04:46:48.220252Z","shell.execute_reply.started":"2022-05-05T04:46:48.057212Z","shell.execute_reply":"2022-05-05T04:46:48.218996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### This table contains all h&m articles with details such as a type of product, a color, a product group and other features.\nArticle data description:\n\narticle_id : A unique identifier of every article.</br>\nproduct_code, prod_name : A unique identifier of every product and its name (not the same).</br>\nproduct_type, product_type_name : The group of product_code and its name</br>\ngraphical_appearance_no, graphical_appearance_name : The group of graphics and its name</br>\ncolour_group_code, colour_group_name : The group of color and its name</br>\nperceived_colour_value_id, perceived_colour_value_name, perceived_colour_master_id, perceived_colour_master_name : The added color info</br>\ndepartment_no, department_name: : A unique identifier of every dep and its name</br>\nindex_code, index_name: : A unique identifier of every index and its name</br>\nindex_group_no, index_group_name: : A group of indeces and its name</br>\nsection_no, section_name: : A unique identifier of every section and its name</br>\ngarment_group_no, garment_group_name: : A unique identifier of every garment and its name</br>\ndetail_desc: : Details</br>","metadata":{"id":"w1RrrohxwRuc"}},{"cell_type":"markdown","source":"#### Some of the columns have -1 values probably referring to missing data","metadata":{"id":"maky1gEUwUIN"}},{"cell_type":"code","source":"print(\"columns having -1 values: \\n\")\ncols_missing_value=[]\nfor i in df_articles.columns:\n    if (-1 in df_articles[i].value_counts()):\n        cols_missing_value.append(i)\nprint(cols_missing_value)       ","metadata":{"id":"E6um8VlKwVtg","outputId":"1b08eef3-3e8a-4148-e69b-058369b7c4cf","execution":{"iopub.status.busy":"2022-05-05T04:46:48.221747Z","iopub.execute_input":"2022-05-05T04:46:48.222053Z","iopub.status.idle":"2022-05-05T04:46:48.562968Z","shell.execute_reply.started":"2022-05-05T04:46:48.222019Z","shell.execute_reply":"2022-05-05T04:46:48.561714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum().sort_values(ascending = False)\n    percent = (data.isnull().sum()/data.isnull().count()*100).sort_values(ascending = False)\n    return pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n\n","metadata":{"id":"fIgFvOhCwZME","execution":{"iopub.status.busy":"2022-05-05T04:46:48.565455Z","iopub.execute_input":"2022-05-05T04:46:48.565862Z","iopub.status.idle":"2022-05-05T04:46:48.572771Z","shell.execute_reply.started":"2022-05-05T04:46:48.565795Z","shell.execute_reply":"2022-05-05T04:46:48.571979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(df_articles).head(7).style.set_properties(**{'background-color': 'rgba(245, 181, 152,.5)'})","metadata":{"id":"T1PRw5J8wcrf","outputId":"d1ffa4b9-5657-48ff-9b3b-c9d6d2d3913c","execution":{"iopub.status.busy":"2022-05-05T04:46:48.573845Z","iopub.execute_input":"2022-05-05T04:46:48.574127Z","iopub.status.idle":"2022-05-05T04:46:49.041105Z","shell.execute_reply.started":"2022-05-05T04:46:48.574084Z","shell.execute_reply":"2022-05-05T04:46:49.039974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### -1 value in all code columns refer to the 'Unknown' category. Therefore keeping -1 values as it is","metadata":{"id":"tUjI4ew5wgVO"}},{"cell_type":"markdown","source":"## Lets Explore some insights about Articles table \n","metadata":{"id":"mucHqUbzxGEq"}},{"cell_type":"markdown","source":"#### i. Ladieswear accounts for a significant part of all dresses. Sportswear has the least portion.\n\n","metadata":{"id":"tebZFgjVxIvC"}},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(15, 7))\nax = sns.histplot(data=df_articles, y='index_name', color='blue')\nax.set_xlabel('count by index name')\nax.set_ylabel('index name')\nplt.show()","metadata":{"id":"KAf_bY86wc8k","outputId":"07bfb708-2025-415e-8cd8-314e83d4a309","execution":{"iopub.status.busy":"2022-05-05T04:46:49.043010Z","iopub.execute_input":"2022-05-05T04:46:49.044048Z","iopub.status.idle":"2022-05-05T04:46:49.464574Z","shell.execute_reply.started":"2022-05-05T04:46:49.043995Z","shell.execute_reply":"2022-05-05T04:46:49.463184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ii. Lets see what ladies buys the most \n\nThe garments grouped by index: As we can see Jersey fancy is the most frequent garment, especially for women and children. The next by number is accessories, many various accessories with low price.\n\n","metadata":{"id":"0EqNHs9ZxRwK"}},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(15, 7))\nax = sns.histplot(data=df_articles, y='garment_group_name', color='blue', hue='index_group_name')\nax.set_xlabel('count by garment group')\nax.set_ylabel('garment group')\nplt.show()","metadata":{"id":"e4CqOfNgwdD7","outputId":"bd14d359-fbcd-4782-8748-f62942aa2e4b","execution":{"iopub.status.busy":"2022-05-05T04:46:49.466073Z","iopub.execute_input":"2022-05-05T04:46:49.466340Z","iopub.status.idle":"2022-05-05T04:46:50.474184Z","shell.execute_reply.started":"2022-05-05T04:46:49.466306Z","shell.execute_reply":"2022-05-05T04:46:50.472916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FEATURE SELECTION & FEATURE ENGINEERING","metadata":{"id":"dNtX-2mc0Edp"}},{"cell_type":"markdown","source":"#### Data splitting is among the most critical steps before preprocessing. It is possible to cause data leakage by dividing the data before processing it, or to overestimate the model evaluation when splitting the data. But Splitting data prior to processing has the huge advantage of ensuring consistency in model performance because unseen data are processed in the same manner as test data.","metadata":{"id":"_m5hRi760X1r"}},{"cell_type":"code","source":"#Split customer sample data (10000 rows) in test and train dataframe\ncustomer_train, customer_test, = train_test_split(customer_sample, test_size=0.3)","metadata":{"id":"opJ25Er30krX","execution":{"iopub.status.busy":"2022-05-05T04:46:50.477742Z","iopub.execute_input":"2022-05-05T04:46:50.478157Z","iopub.status.idle":"2022-05-05T04:46:50.490237Z","shell.execute_reply.started":"2022-05-05T04:46:50.478109Z","shell.execute_reply":"2022-05-05T04:46:50.488818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customer_train.head()\ncustomer_test.shape \ncustomer_train.shape ","metadata":{"id":"eoa7nKnC0t7T","outputId":"22f26edb-5539-4bb0-c1c6-1eab61347db7","execution":{"iopub.status.busy":"2022-05-05T04:46:50.491671Z","iopub.execute_input":"2022-05-05T04:46:50.491927Z","iopub.status.idle":"2022-05-05T04:46:50.506666Z","shell.execute_reply.started":"2022-05-05T04:46:50.491892Z","shell.execute_reply":"2022-05-05T04:46:50.505703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data frame df_articles has been cleaned up by removing extra columns. We need to drop a couple of columns from the df_articles data frame since it is extremely large, and we won't be evaluating the machine learning model with those columns.","metadata":{"id":"GjBsefM9yTsq"}},{"cell_type":"code","source":"# remove unwanted columns from articles df and place it into a new df\nclean_df_articles = df_articles.drop(columns=['prod_name', 'product_type_name', 'graphical_appearance_name',\n                                       'colour_group_name', 'perceived_colour_value_name', 'perceived_colour_master_name', 'department_name', \n                                'perceived_colour_value_id', 'perceived_colour_master_id',\n                                       'index_name', 'index_group_name', 'section_name', 'garment_group_name'])","metadata":{"id":"NpoYcR1E45e9","execution":{"iopub.status.busy":"2022-05-05T04:46:50.508386Z","iopub.execute_input":"2022-05-05T04:46:50.508615Z","iopub.status.idle":"2022-05-05T04:46:50.528767Z","shell.execute_reply.started":"2022-05-05T04:46:50.508583Z","shell.execute_reply":"2022-05-05T04:46:50.527350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_df_articles.info()","metadata":{"id":"pH4REVUuygFl","outputId":"5e55a671-68ef-4300-8b9e-9c9c409b6f45","execution":{"iopub.status.busy":"2022-05-05T04:46:50.530875Z","iopub.execute_input":"2022-05-05T04:46:50.531340Z","iopub.status.idle":"2022-05-05T04:46:50.588313Z","shell.execute_reply.started":"2022-05-05T04:46:50.531290Z","shell.execute_reply":"2022-05-05T04:46:50.587078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### From the above cell, we can see that unwated columns have been removed. Additionally, we convert the data types for the article column to categorical to reduce processing time.","metadata":{"id":"eMekiWCyyxaC"}},{"cell_type":"code","source":"clean_df_articles['product_code'] = clean_df_articles.product_code.astype('category')\nclean_df_articles['product_type_no'] = clean_df_articles.product_type_no.astype('category')\nclean_df_articles['product_group_name'] = clean_df_articles.product_group_name.astype('category')\nclean_df_articles['graphical_appearance_no'] = clean_df_articles.graphical_appearance_no.astype('category')\nclean_df_articles['colour_group_code'] = clean_df_articles.colour_group_code.astype('category')\nclean_df_articles['department_no'] = clean_df_articles.department_no.astype('category')\nclean_df_articles['index_code'] = clean_df_articles.product_type_no.astype('category')\nclean_df_articles['index_group_no'] = clean_df_articles.index_group_no.astype('category')\nclean_df_articles['section_no'] = clean_df_articles.section_no.astype('category')\nclean_df_articles['garment_group_no'] = clean_df_articles.garment_group_no.astype('category')\nclean_df_articles['detail_desc'] = clean_df_articles.detail_desc.astype('category')","metadata":{"id":"3Sa5vqyU5lAK","execution":{"iopub.status.busy":"2022-05-05T04:46:50.590088Z","iopub.execute_input":"2022-05-05T04:46:50.590757Z","iopub.status.idle":"2022-05-05T04:46:50.795689Z","shell.execute_reply.started":"2022-05-05T04:46:50.590716Z","shell.execute_reply":"2022-05-05T04:46:50.794271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_df_articles.info()","metadata":{"id":"ZE1PL_4t5hWw","outputId":"d381abcb-6cba-44ae-d735-eea8c3c21c2a","execution":{"iopub.status.busy":"2022-05-05T04:46:50.798233Z","iopub.execute_input":"2022-05-05T04:46:50.798619Z","iopub.status.idle":"2022-05-05T04:46:50.873878Z","shell.execute_reply.started":"2022-05-05T04:46:50.798569Z","shell.execute_reply":"2022-05-05T04:46:50.872547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge df_article and df_transastion table into a new df called article_transactions","metadata":{"id":"Lo_F01E31qAj"}},{"cell_type":"code","source":"article_transactions = df_transactions.merge(clean_df_articles, on='article_id', how='left')","metadata":{"id":"HHKReDcM6J2B","execution":{"iopub.status.busy":"2022-05-05T04:46:50.875520Z","iopub.execute_input":"2022-05-05T04:46:50.875794Z","iopub.status.idle":"2022-05-05T04:46:59.713561Z","shell.execute_reply.started":"2022-05-05T04:46:50.875756Z","shell.execute_reply":"2022-05-05T04:46:59.712197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_transactions.info()","metadata":{"id":"SrweSl4J9vp6","outputId":"75706e23-1132-4cda-d134-234ae3114b4b","execution":{"iopub.status.busy":"2022-05-05T04:46:59.721411Z","iopub.execute_input":"2022-05-05T04:46:59.722445Z","iopub.status.idle":"2022-05-05T04:46:59.745307Z","shell.execute_reply.started":"2022-05-05T04:46:59.722382Z","shell.execute_reply":"2022-05-05T04:46:59.743917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_transactions.head(2)","metadata":{"id":"uweln1jG1j3o","outputId":"19784fd3-2270-4efb-a9f4-c5601acca9b7","execution":{"iopub.status.busy":"2022-05-05T04:46:59.746764Z","iopub.execute_input":"2022-05-05T04:46:59.747119Z","iopub.status.idle":"2022-05-05T04:46:59.777147Z","shell.execute_reply.started":"2022-05-05T04:46:59.747083Z","shell.execute_reply":"2022-05-05T04:46:59.776467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"YuoFvHeT1qAi"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge customer training and customer test dataframe which we created earlier with article_transactions df to create a proper Train & Test Df which we will use for our testing and prediction.","metadata":{"id":"Wn9QIqQz1qbi"}},{"cell_type":"code","source":"# Merge customer_train,test with article_transactions to get train & test dataframe\ntrain = customer_train.merge(article_transactions, on='customer_id', how='inner') \ntest = customer_test.merge(article_transactions, on='customer_id', how='inner')","metadata":{"id":"wsZueHZC95uI","execution":{"iopub.status.busy":"2022-05-05T04:46:59.778310Z","iopub.execute_input":"2022-05-05T04:46:59.779341Z","iopub.status.idle":"2022-05-05T04:47:28.841311Z","shell.execute_reply.started":"2022-05-05T04:46:59.779266Z","shell.execute_reply":"2022-05-05T04:47:28.840649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()\ntest.info()","metadata":{"id":"ke_JxI0e95xX","outputId":"8fbad5c6-4daf-4d44-b3f1-78172b89d1a6","execution":{"iopub.status.busy":"2022-05-05T04:47:28.842515Z","iopub.execute_input":"2022-05-05T04:47:28.842821Z","iopub.status.idle":"2022-05-05T04:47:29.134706Z","shell.execute_reply.started":"2022-05-05T04:47:28.842790Z","shell.execute_reply":"2022-05-05T04:47:29.133460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(-2)","metadata":{"id":"vb0EiIJj9yaa","outputId":"09441461-d466-4436-bd4e-41857f291a34","execution":{"iopub.status.busy":"2022-05-05T04:47:29.135999Z","iopub.execute_input":"2022-05-05T04:47:29.136240Z","iopub.status.idle":"2022-05-05T04:47:29.275635Z","shell.execute_reply.started":"2022-05-05T04:47:29.136207Z","shell.execute_reply":"2022-05-05T04:47:29.274643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert Column t_dat in both train & test df to proper datetime format","metadata":{"id":"9xBbHo0n2KF5"}},{"cell_type":"code","source":"#Convert Column t_dat in both train & test df to proper datetime format\ntrain.t_dat = pd.to_datetime(train.t_dat)\ntest.t_dat = pd.to_datetime(test.t_dat)","metadata":{"id":"61ujlOQr_5WN","execution":{"iopub.status.busy":"2022-05-05T04:47:29.276785Z","iopub.execute_input":"2022-05-05T04:47:29.277842Z","iopub.status.idle":"2022-05-05T04:47:29.419169Z","shell.execute_reply.started":"2022-05-05T04:47:29.277784Z","shell.execute_reply":"2022-05-05T04:47:29.417351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Sort the customer_id values from test & train df\ntrain = train.sort_values(['customer_id', 't_dat'])\ntest = test.sort_values(['customer_id', 't_dat'])","metadata":{"id":"Tz7ugySj_5Y5","execution":{"iopub.status.busy":"2022-05-05T04:47:29.421021Z","iopub.execute_input":"2022-05-05T04:47:29.421507Z","iopub.status.idle":"2022-05-05T04:47:29.693363Z","shell.execute_reply.started":"2022-05-05T04:47:29.421451Z","shell.execute_reply":"2022-05-05T04:47:29.692260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get last 12 item buys of all the customers","metadata":{"id":"OZQwOt2a2lLj"}},{"cell_type":"code","source":"train_group = train.groupby('customer_id', observed=True).tail(12).index #List of the 12 last index\ntest_group = test.groupby('customer_id', observed=True).tail(12).index","metadata":{"id":"ru9wom8x_5by","execution":{"iopub.status.busy":"2022-05-05T04:47:29.694900Z","iopub.execute_input":"2022-05-05T04:47:29.695376Z","iopub.status.idle":"2022-05-05T04:47:29.898191Z","shell.execute_reply.started":"2022-05-05T04:47:29.695330Z","shell.execute_reply":"2022-05-05T04:47:29.897075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#replaces everything up to groupby in next cell\ntrain.loc[train_group,:] \ntest.loc[test_group,:]","metadata":{"id":"kOiiYyXN_5ed","outputId":"c291fef9-86f7-4d25-d220-401129a5689e","execution":{"iopub.status.busy":"2022-05-05T04:47:29.899752Z","iopub.execute_input":"2022-05-05T04:47:29.900141Z","iopub.status.idle":"2022-05-05T04:47:30.001458Z","shell.execute_reply.started":"2022-05-05T04:47:29.900102Z","shell.execute_reply":"2022-05-05T04:47:30.000021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We drop unwanted columns from train and test to get dataframes with customers and their last 12 purchases only.\nThis will give us a y_truth value which states real purchases by customers.","metadata":{"id":"RlvN1sjz6PIV"}},{"cell_type":"code","source":"#train.drop(index=train_group, inplace=True)\n#test.drop(index=test_group, inplace=True)","metadata":{"id":"6-gz-O_j5bOE","execution":{"iopub.status.busy":"2022-05-05T04:47:30.003647Z","iopub.execute_input":"2022-05-05T04:47:30.004508Z","iopub.status.idle":"2022-05-05T04:47:30.009428Z","shell.execute_reply.started":"2022-05-05T04:47:30.004458Z","shell.execute_reply":"2022-05-05T04:47:30.007896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train.loc[train_group,:].groupby('customer_id', observed=True)['article_id'].apply(lambda x: x.tolist())\ny_test =  test.loc[test_group,:].groupby('customer_id', observed=True)['article_id'].apply(lambda x: x.tolist())","metadata":{"id":"_Uq1kjDY_5hD","execution":{"iopub.status.busy":"2022-05-05T04:47:30.010979Z","iopub.execute_input":"2022-05-05T04:47:30.012028Z","iopub.status.idle":"2022-05-05T04:47:30.340806Z","shell.execute_reply.started":"2022-05-05T04:47:30.011963Z","shell.execute_reply":"2022-05-05T04:47:30.339974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.head(100)\ny_test.head(100)","metadata":{"id":"NGCFyxWHihs5","outputId":"edbfd2b8-ae07-4012-a1e3-68a0f3dab4a5","execution":{"iopub.status.busy":"2022-05-05T04:47:30.342530Z","iopub.execute_input":"2022-05-05T04:47:30.342864Z","iopub.status.idle":"2022-05-05T04:47:30.355485Z","shell.execute_reply.started":"2022-05-05T04:47:30.342827Z","shell.execute_reply":"2022-05-05T04:47:30.354428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create data frame for y_train & y_test to get value for just 1 customer\nuno_y_train = y_train.apply(lambda x: x[1])\nuno_y_test = y_test.apply(lambda x: x[1])","metadata":{"id":"6dtdbJfBOC3g","execution":{"iopub.status.busy":"2022-05-05T04:47:30.356974Z","iopub.execute_input":"2022-05-05T04:47:30.357299Z","iopub.status.idle":"2022-05-05T04:47:30.379659Z","shell.execute_reply.started":"2022-05-05T04:47:30.357253Z","shell.execute_reply":"2022-05-05T04:47:30.378473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uno_y_train.head()","metadata":{"id":"JcQQnxn1PlTl","outputId":"4cfef562-2164-481b-ad6b-864766082d27","execution":{"iopub.status.busy":"2022-05-05T04:47:30.381105Z","iopub.execute_input":"2022-05-05T04:47:30.382274Z","iopub.status.idle":"2022-05-05T04:47:30.390706Z","shell.execute_reply.started":"2022-05-05T04:47:30.382203Z","shell.execute_reply":"2022-05-05T04:47:30.389914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uno_y_test.head()","metadata":{"id":"3kTG8OFNNrjF","outputId":"eb078f8b-2551-44a7-c647-d4d3b5a0eb28","execution":{"iopub.status.busy":"2022-05-05T04:47:30.392070Z","iopub.execute_input":"2022-05-05T04:47:30.393087Z","iopub.status.idle":"2022-05-05T04:47:30.407522Z","shell.execute_reply.started":"2022-05-05T04:47:30.393039Z","shell.execute_reply":"2022-05-05T04:47:30.405444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to get X_train & x_test data frames. \nThis function will helpfull to finding the features values which will be used for finding accuracy.\n","metadata":{"id":"KLxswkt-Ayp2"}},{"cell_type":"code","source":"def function_features (customers):\n        #here we will fetch all the required columns which we will use as features\n        features_rows = {'FN' : customers['FN'].iloc[0],\n                    'Active' : customers['Active'].iloc[0],\n                    'club_member_status' : customers['club_member_status'].iloc[0],\n                    'fashion_news_frequency' :customers['fashion_news_frequency'].iloc[0],\n                    'age'  : customers['age'].iloc[0],\n                    'postal_code' : customers['postal_code'].iloc[0]} \n        features_rows['bought_items'] = customers.shape[0] #\n        return pd.Series(features_rows) \n        \n","metadata":{"id":"vkP1IYNMAgjz","execution":{"iopub.status.busy":"2022-05-05T04:47:30.408788Z","iopub.execute_input":"2022-05-05T04:47:30.409068Z","iopub.status.idle":"2022-05-05T04:47:30.417455Z","shell.execute_reply.started":"2022-05-05T04:47:30.409035Z","shell.execute_reply":"2022-05-05T04:47:30.416747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Getting x_train & x_test","metadata":{"id":"CzbgCbcgCiF7"}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\ntqdm.pandas()\n\nx_train = train.groupby('customer_id', observed=True).progress_apply(function_features) #Apply feat_gen to entire Full_train\n\nx_test  = test.groupby('customer_id', observed=True).progress_apply(function_features) ","metadata":{"id":"ZigTwkhiAgrq","outputId":"ab7b345a-5657-45b9-f621-3cd8719e3329","execution":{"iopub.status.busy":"2022-05-05T04:47:30.418482Z","iopub.execute_input":"2022-05-05T04:47:30.419110Z","iopub.status.idle":"2022-05-05T04:47:40.666132Z","shell.execute_reply.started":"2022-05-05T04:47:30.419068Z","shell.execute_reply":"2022-05-05T04:47:40.664759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.head()","metadata":{"id":"J9FgiFLdAguC","outputId":"815e5ccc-fe66-4034-c773-201d91429b58","execution":{"iopub.status.busy":"2022-05-05T04:47:54.796744Z","iopub.execute_input":"2022-05-05T04:47:54.798259Z","iopub.status.idle":"2022-05-05T04:47:54.816167Z","shell.execute_reply.started":"2022-05-05T04:47:54.798171Z","shell.execute_reply":"2022-05-05T04:47:54.815272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.head()","metadata":{"id":"4xGjwiwGAgwh","outputId":"f9e8bc31-a4db-4d59-c75b-c4299629359d","execution":{"iopub.status.busy":"2022-05-05T04:47:57.436329Z","iopub.execute_input":"2022-05-05T04:47:57.436792Z","iopub.status.idle":"2022-05-05T04:47:57.456504Z","shell.execute_reply.started":"2022-05-05T04:47:57.436751Z","shell.execute_reply":"2022-05-05T04:47:57.454760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now we get the prediction column (y_prediction) for training data\n","metadata":{"id":"PCmDRa5vDWeL"}},{"cell_type":"code","source":"train_prediction = train.groupby([\"customer_id\"])[\"article_id\"].agg(lambda x: str(x.values[0:12])[1:-1]).reset_index()","metadata":{"id":"NUD6b0WBAgy1","execution":{"iopub.status.busy":"2022-05-05T04:48:03.467282Z","iopub.execute_input":"2022-05-05T04:48:03.467603Z","iopub.status.idle":"2022-05-05T04:48:04.303264Z","shell.execute_reply.started":"2022-05-05T04:48:03.467566Z","shell.execute_reply":"2022-05-05T04:48:04.302022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets define a function to get prediction values by splitting all items values.","metadata":{"id":"-Czn_E5hDwsH"}},{"cell_type":"code","source":"def articles_padding(x):\n    if x:\n        xl = x.split()\n        x = []\n        for xi in xl:\n            x.append(\"0\"+xi)\n        dimm_x = len(x)\n        if dimm_x < 12:\n            x.extend(art_list[:12-dimm_x])\n        return(\" \".join(x))","metadata":{"id":"k_3PpZ97Ag1L","execution":{"iopub.status.busy":"2022-05-05T04:48:06.016004Z","iopub.execute_input":"2022-05-05T04:48:06.016333Z","iopub.status.idle":"2022-05-05T04:48:06.023336Z","shell.execute_reply.started":"2022-05-05T04:48:06.016297Z","shell.execute_reply":"2022-05-05T04:48:06.022423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_prediction[\"article_id\"] = train_prediction[\"article_id\"].apply(lambda x: articles_padding(x))","metadata":{"id":"hvlP34zbAg3j","execution":{"iopub.status.busy":"2022-05-05T04:48:08.072586Z","iopub.execute_input":"2022-05-05T04:48:08.073189Z","iopub.status.idle":"2022-05-05T04:48:08.116828Z","shell.execute_reply.started":"2022-05-05T04:48:08.073136Z","shell.execute_reply":"2022-05-05T04:48:08.116114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Temporary predictied articles ( we will get proper prediction later.)","metadata":{"id":"MVrAGfFNEj_R"}},{"cell_type":"code","source":"train_prediction.head()","metadata":{"id":"LbTEzIJkAg6m","outputId":"b8e1fa62-e1ac-4cca-9ff4-9ecfe91f4356","execution":{"iopub.status.busy":"2022-05-05T04:48:10.417802Z","iopub.execute_input":"2022-05-05T04:48:10.418125Z","iopub.status.idle":"2022-05-05T04:48:10.430533Z","shell.execute_reply.started":"2022-05-05T04:48:10.418089Z","shell.execute_reply":"2022-05-05T04:48:10.429109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now we get the prediction column (y_prediction) for test data\n","metadata":{"id":"C9mpJE0YFDGS"}},{"cell_type":"code","source":"test_prediction = test.groupby([\"customer_id\"])[\"article_id\"].agg(lambda x: str(x.values[0:12])[1:-1]).reset_index()","metadata":{"id":"iA0oZjd3E62y","execution":{"iopub.status.busy":"2022-05-05T04:48:12.236989Z","iopub.execute_input":"2022-05-05T04:48:12.238271Z","iopub.status.idle":"2022-05-05T04:48:12.579300Z","shell.execute_reply.started":"2022-05-05T04:48:12.238201Z","shell.execute_reply":"2022-05-05T04:48:12.577983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def articles_padding_test(x):\n    if x:\n        xl = x.split()\n        x = []\n        for xi in xl:\n            x.append(\"0\"+xi)\n        dimm_x = len(x)\n        if dimm_x < 12:\n            x.extend(art_list[:12-dimm_x])\n        return(\" \".join(x))","metadata":{"id":"ux3Xr-VrE7KJ","execution":{"iopub.status.busy":"2022-05-05T04:48:13.723807Z","iopub.execute_input":"2022-05-05T04:48:13.724202Z","iopub.status.idle":"2022-05-05T04:48:13.732213Z","shell.execute_reply.started":"2022-05-05T04:48:13.724160Z","shell.execute_reply":"2022-05-05T04:48:13.730929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prediction[\"article_id\"] = test_prediction[\"article_id\"].apply(lambda x: articles_padding_test(x))","metadata":{"id":"MwhyX3TrE7OY","execution":{"iopub.status.busy":"2022-05-05T04:48:15.762290Z","iopub.execute_input":"2022-05-05T04:48:15.762625Z","iopub.status.idle":"2022-05-05T04:48:15.786390Z","shell.execute_reply.started":"2022-05-05T04:48:15.762587Z","shell.execute_reply":"2022-05-05T04:48:15.785396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prediction.head()","metadata":{"id":"HGI6VAQ4FH7b","outputId":"363a7a42-055e-418e-8ea4-156efc3bcba3","execution":{"iopub.status.busy":"2022-05-05T04:48:17.596874Z","iopub.execute_input":"2022-05-05T04:48:17.597235Z","iopub.status.idle":"2022-05-05T04:48:17.608860Z","shell.execute_reply.started":"2022-05-05T04:48:17.597193Z","shell.execute_reply":"2022-05-05T04:48:17.607996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Machine Learning Algorithm to Find accuracy and Predcition of the training and test models.","metadata":{"id":"PSJTV7_gErCu"}},{"cell_type":"markdown","source":"Before we proceed with ml part , we will need to convert our values in x_train & x_test to one hot encoding because currently we have all the data types as categorical which will be needed to converted to int or float to proceed with predcting accuracy","metadata":{"id":"wp5V-dc7FX-4"}},{"cell_type":"code","source":"# remove na's in x_train\nx_train.fillna(0)","metadata":{"id":"RqRm5Jxf8DGS","outputId":"b60f59be-ae59-42d2-82f0-af01daaeb921","execution":{"iopub.status.busy":"2022-05-05T04:48:20.029860Z","iopub.execute_input":"2022-05-05T04:48:20.030657Z","iopub.status.idle":"2022-05-05T04:48:20.060656Z","shell.execute_reply.started":"2022-05-05T04:48:20.030602Z","shell.execute_reply":"2022-05-05T04:48:20.059813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove na's in x_test\nx_test.fillna(0)","metadata":{"id":"D00Pv5x08DI3","outputId":"6569204b-c24f-4b95-c5ae-91f6411badd1","execution":{"iopub.status.busy":"2022-05-05T04:48:22.049097Z","iopub.execute_input":"2022-05-05T04:48:22.049554Z","iopub.status.idle":"2022-05-05T04:48:22.079531Z","shell.execute_reply.started":"2022-05-05T04:48:22.049520Z","shell.execute_reply":"2022-05-05T04:48:22.078132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LETS CONVERT 'FN', 'Active', 'club_member_status', 'fashion_news_frequency' columns (these are categorical columns) to binary using one hot encoding in x_train and x_test","metadata":{"id":"fWevH12YGMYd"}},{"cell_type":"code","source":"# applying the one hot encoding  to our X train and  X test dataframe  for 'FN', 'Active', 'club_member_status', 'fashion_news_frequency' columns\nx_train_encoded = pd.get_dummies(x_train, columns=['FN', 'Active', 'club_member_status', 'fashion_news_frequency'])\nx_test_encoded = pd.get_dummies(x_test, columns=['FN', 'Active', 'club_member_status', 'fashion_news_frequency'])","metadata":{"id":"8eID61M38DLS","execution":{"iopub.status.busy":"2022-05-05T04:48:25.279282Z","iopub.execute_input":"2022-05-05T04:48:25.279592Z","iopub.status.idle":"2022-05-05T04:48:25.316185Z","shell.execute_reply.started":"2022-05-05T04:48:25.279554Z","shell.execute_reply":"2022-05-05T04:48:25.314994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lets drop unwated column (postal_code, as it is not needed)\nx_train_encoded.drop(['postal_code'], axis=1, inplace=True)\nx_test_encoded.drop(['postal_code'], axis=1, inplace=True)","metadata":{"id":"2uvJMw-GGLyT","execution":{"iopub.status.busy":"2022-05-05T04:48:29.819126Z","iopub.execute_input":"2022-05-05T04:48:29.819453Z","iopub.status.idle":"2022-05-05T04:48:29.830291Z","shell.execute_reply.started":"2022-05-05T04:48:29.819415Z","shell.execute_reply":"2022-05-05T04:48:29.828720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nx_train_encoded.head()\n","metadata":{"id":"ef7vySuTGL4D","outputId":"7b7c45d6-b68b-497b-c9b6-e1a2ebdf79fa","execution":{"iopub.status.busy":"2022-05-05T04:48:31.248027Z","iopub.execute_input":"2022-05-05T04:48:31.248541Z","iopub.status.idle":"2022-05-05T04:48:31.266757Z","shell.execute_reply.started":"2022-05-05T04:48:31.248505Z","shell.execute_reply":"2022-05-05T04:48:31.265536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test_encoded.head()","metadata":{"id":"HIEVtV8aGL9F","outputId":"c8c40bd9-db68-4618-b9c0-bcad944c970e","execution":{"iopub.status.busy":"2022-05-05T04:48:32.697324Z","iopub.execute_input":"2022-05-05T04:48:32.697791Z","iopub.status.idle":"2022-05-05T04:48:32.712242Z","shell.execute_reply.started":"2022-05-05T04:48:32.697747Z","shell.execute_reply":"2022-05-05T04:48:32.711193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that all categorical columns are now converted to binary 1's & 0's to make our job easier to get accuracy","metadata":{"id":"rYotvzmCHKW5"}},{"cell_type":"markdown","source":"## Now lets find accuracy using customers data which we did to get a value","metadata":{"id":"3WuYo3w_HXqb"}},{"cell_type":"markdown","source":"# i. DECISION TREE","metadata":{"id":"JuZJvQOxW3lw"}},{"cell_type":"code","source":"\nfrom sklearn.tree import DecisionTreeClassifier # Import Decision Tree Classifier\nfrom sklearn.model_selection import train_test_split # Import train_test_split function\nfrom sklearn import metrics #Import scikit-learn metrics module for accuracy calculation","metadata":{"id":"PfQJzt2kIzQm","execution":{"iopub.status.busy":"2022-05-05T04:48:35.453489Z","iopub.execute_input":"2022-05-05T04:48:35.453840Z","iopub.status.idle":"2022-05-05T04:48:35.458542Z","shell.execute_reply.started":"2022-05-05T04:48:35.453798Z","shell.execute_reply":"2022-05-05T04:48:35.457701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Decision tree classifer object\nclf = DecisionTreeClassifier()\n\n# Train Decision Tree Classifer\nclf = clf.fit(x_train_encoded,uno_y_train)\n\n#Predict the response for test dataset\ny_pred = clf.predict(x_train_encoded)","metadata":{"id":"2Qd7TXo8IzTa","execution":{"iopub.status.busy":"2022-05-05T04:48:36.783529Z","iopub.execute_input":"2022-05-05T04:48:36.783866Z","iopub.status.idle":"2022-05-05T04:48:37.486211Z","shell.execute_reply.started":"2022-05-05T04:48:36.783831Z","shell.execute_reply":"2022-05-05T04:48:37.485146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_train)","metadata":{"id":"h1i2JpCGM1sM","outputId":"3760e7ac-1a97-4836-dd15-b33c58ee5571","execution":{"iopub.status.busy":"2022-05-05T04:48:38.481566Z","iopub.execute_input":"2022-05-05T04:48:38.481861Z","iopub.status.idle":"2022-05-05T04:48:38.492212Z","shell.execute_reply.started":"2022-05-05T04:48:38.481826Z","shell.execute_reply":"2022-05-05T04:48:38.490739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred)","metadata":{"id":"uK1TX9AOSvYW","outputId":"9b8928d3-930f-4cfe-95b5-9132bd2f7518","execution":{"iopub.status.busy":"2022-05-05T04:48:40.157027Z","iopub.execute_input":"2022-05-05T04:48:40.157357Z","iopub.status.idle":"2022-05-05T04:48:40.163736Z","shell.execute_reply.started":"2022-05-05T04:48:40.157318Z","shell.execute_reply":"2022-05-05T04:48:40.162900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ** WE GET A ACCURACY OF 60 % using Decison Tree WHICH TELLS US THAT OUR PREDICTION WAS CORRECT**\n\n---\n\n","metadata":{"id":"rtYWyfmDXG1z"}},{"cell_type":"code","source":"print(\"Accuracy:\",metrics.accuracy_score(uno_y_train, y_pred))","metadata":{"id":"cLYu4AkiSxnF","outputId":"37ff74ad-4704-4e23-fe59-05e42a72eef8","execution":{"iopub.status.busy":"2022-05-05T04:48:42.366135Z","iopub.execute_input":"2022-05-05T04:48:42.366624Z","iopub.status.idle":"2022-05-05T04:48:42.374030Z","shell.execute_reply.started":"2022-05-05T04:48:42.366585Z","shell.execute_reply":"2022-05-05T04:48:42.373219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"j2BlA3zPXXpt"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ii. KNN Algorithm","metadata":{"id":"r_0AqM1YXYQm"}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\n\nX = x_train_encoded\ny = uno_y_train\n\nKnnc = KNeighborsClassifier(n_neighbors=5)\nKnnc.fit(X,y)\ny_pred = Knnc.predict(x_train_encoded)\n\nprint(accuracy_score(uno_y_train, y_pred)) \n","metadata":{"id":"1_F1o--ELM9l","outputId":"5e2fcf5c-1d32-41b9-f1a9-a92178731eaa","execution":{"iopub.status.busy":"2022-05-05T04:48:46.073199Z","iopub.execute_input":"2022-05-05T04:48:46.074374Z","iopub.status.idle":"2022-05-05T04:48:46.402729Z","shell.execute_reply.started":"2022-05-05T04:48:46.074306Z","shell.execute_reply":"2022-05-05T04:48:46.401441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets Get single predictions  for all customer using following function","metadata":{"id":"9qF2HFw6ZWeW"}},{"cell_type":"code","source":"#allx = pd.concat([x_train_encoded, x_test_encoded]) \n#average_customer = allx.mean(axis=0).to_frame().T\n#missing_new = df_customers['customer_id'][~df_customers['customer_id'].isin(allx.index)]\n#submissions = clf.predict(allx)\n#customer_ppred = clf.predict(average_customer)","metadata":{"id":"yDQOUJuiIOYZ","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submissions = pd.DataFrame({'customer_id': allx.index, 'prediction' : submissions})\n#submissions = pd.concat([submissions, pd.DataFrame({'customer_id': missing_new, 'prediction': np.repeat(customer_ppred, len(missing_new))})])","metadata":{"id":"0As2trOHaCin","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submissions.sort_index(inplace=True)","metadata":{"id":"nhTjIF22djWB","execution":{"iopub.status.busy":"2022-05-05T04:47:43.674899Z","iopub.status.idle":"2022-05-05T04:47:43.675352Z","shell.execute_reply.started":"2022-05-05T04:47:43.675114Z","shell.execute_reply":"2022-05-05T04:47:43.675137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"NnD5xcuucjhi"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets Get 12 prediction for all customers using following function","metadata":{"id":"PxBeWaP-eb_N"}},{"cell_type":"code","source":"pred_df = df_transactions.groupby([\"customer_id\"])[\"article_id\"].agg(lambda x: str(x.values[0:12])[1:-1]).reset_index()","metadata":{"id":"B78rk0Tbef38","execution":{"iopub.status.busy":"2022-05-05T04:53:47.107460Z","iopub.execute_input":"2022-05-05T04:53:47.107752Z","iopub.status.idle":"2022-05-05T04:56:01.868526Z","shell.execute_reply.started":"2022-05-05T04:53:47.107717Z","shell.execute_reply":"2022-05-05T04:56:01.867585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find customers with purchases on last days\nlast_date = df_transactions.t_dat.max()\nprint(df_transactions.loc[df_transactions.t_dat==last_date].shape)\n\n\n# find most frequent items\nmost_frequent_articles = list(df_transactions.loc[df_transactions.t_dat==last_date].article_id.value_counts()[0:12].index)\nart_list = []\nfor art in most_frequent_articles:\n    art = \"0\"+str(art)\n    art_list.append(art)\nart_str = \" \".join(art_list)\nprint(\"Frequent articles bought recently:\", art_str, end=\"\\n\")","metadata":{"id":"5LlYwcwNfSoT","outputId":"b4cd9d41-0da5-40d8-b8ec-9dadd047c51e","execution":{"iopub.status.busy":"2022-05-05T04:56:01.870329Z","iopub.execute_input":"2022-05-05T04:56:01.871391Z","iopub.status.idle":"2022-05-05T04:56:15.596360Z","shell.execute_reply.started":"2022-05-05T04:56:01.871352Z","shell.execute_reply":"2022-05-05T04:56:15.595041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def padding_articles_prediction(x):\n    if x:\n        xl = x.split()\n        x = []\n        for xi in xl:\n            x.append(\"0\"+xi)\n        dimm_x = len(x)\n        if dimm_x < 12:\n            x.extend(art_list[:12-dimm_x])\n        return(\" \".join(x))","metadata":{"id":"SP4edQq_ejk3","execution":{"iopub.status.busy":"2022-05-05T04:56:15.598459Z","iopub.execute_input":"2022-05-05T04:56:15.598756Z","iopub.status.idle":"2022-05-05T04:56:15.606479Z","shell.execute_reply.started":"2022-05-05T04:56:15.598720Z","shell.execute_reply":"2022-05-05T04:56:15.605043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df[\"article_id\"] = pred_df[\"article_id\"].apply(lambda x: padding_articles_prediction(x))\n","metadata":{"id":"Vf2zcI6YfAYI","execution":{"iopub.status.busy":"2022-05-05T04:56:15.609434Z","iopub.execute_input":"2022-05-05T04:56:15.610375Z","iopub.status.idle":"2022-05-05T04:56:21.300986Z","shell.execute_reply.started":"2022-05-05T04:56:15.610292Z","shell.execute_reply":"2022-05-05T04:56:21.299675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace sample submission files with our predicted values\ndf_submission = pred_df.merge(df_sample_submission[[\"customer_id\"]], how=\"right\")\ndf_submission.columns = [\"customer_id\", \"prediction\"]\ndf_submission.head().style.set_properties(**{'background-color': 'rgba(184,230,194,.5)'})","metadata":{"id":"0k2M7xCMfGQk","outputId":"066ec9d9-b2d4-4c88-f424-cbae946001bf","execution":{"iopub.status.busy":"2022-05-05T04:56:21.303050Z","iopub.execute_input":"2022-05-05T04:56:21.303431Z","iopub.status.idle":"2022-05-05T04:56:23.969476Z","shell.execute_reply.started":"2022-05-05T04:56:21.303390Z","shell.execute_reply":"2022-05-05T04:56:23.968352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv',index=False)","metadata":{"id":"JlvpZ1l0f2mh","execution":{"iopub.status.busy":"2022-05-05T04:56:26.269752Z","iopub.execute_input":"2022-05-05T04:56:26.270079Z","iopub.status.idle":"2022-05-05T04:56:36.443800Z","shell.execute_reply.started":"2022-05-05T04:56:26.270043Z","shell.execute_reply":"2022-05-05T04:56:36.442813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}