{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:29:02.704196Z","iopub.execute_input":"2022-03-30T03:29:02.704575Z","iopub.status.idle":"2022-03-30T03:29:02.709764Z","shell.execute_reply.started":"2022-03-30T03:29:02.704533Z","shell.execute_reply":"2022-03-30T03:29:02.708553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"code","source":"# load dataset\narticles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:26:33.262061Z","iopub.execute_input":"2022-03-30T03:26:33.262915Z","iopub.status.idle":"2022-03-30T03:27:51.619209Z","shell.execute_reply.started":"2022-03-30T03:26:33.262865Z","shell.execute_reply":"2022-03-30T03:27:51.617684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.info()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:27:51.635098Z","iopub.execute_input":"2022-03-30T03:27:51.635919Z","iopub.status.idle":"2022-03-30T03:27:51.664106Z","shell.execute_reply.started":"2022-03-30T03:27:51.635874Z","shell.execute_reply":"2022-03-30T03:27:51.662863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:27:51.666179Z","iopub.execute_input":"2022-03-30T03:27:51.666709Z","iopub.status.idle":"2022-03-30T03:27:51.696499Z","shell.execute_reply.started":"2022-03-30T03:27:51.666654Z","shell.execute_reply":"2022-03-30T03:27:51.694872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions['t_dat'].max()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:27:51.699147Z","iopub.execute_input":"2022-03-30T03:27:51.699668Z","iopub.status.idle":"2022-03-30T03:27:56.045676Z","shell.execute_reply.started":"2022-03-30T03:27:51.699631Z","shell.execute_reply":"2022-03-30T03:27:56.044596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 10000 samples of tansactions\ndf_sample = transactions.sample(n=10000)\ndf_sample.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:27:56.047608Z","iopub.execute_input":"2022-03-30T03:27:56.047870Z","iopub.status.idle":"2022-03-30T03:27:59.286154Z","shell.execute_reply.started":"2022-03-30T03:27:56.047839Z","shell.execute_reply":"2022-03-30T03:27:59.285020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one month sample of data\n# df_sample = transactions[transactions['t_dat'] > '2020-09-10']\n\n# df_sample.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:27:59.288011Z","iopub.execute_input":"2022-03-30T03:27:59.289087Z","iopub.status.idle":"2022-03-30T03:27:59.292716Z","shell.execute_reply.started":"2022-03-30T03:27:59.289025Z","shell.execute_reply":"2022-03-30T03:27:59.291729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#join customer data\ndf_sample = pd.merge(df_sample, customers, on='customer_id')\n\n#join article data\n# df_sample =pd.merge(df_sample, articles, on='article_id')","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:27:59.294712Z","iopub.execute_input":"2022-03-30T03:27:59.295173Z","iopub.status.idle":"2022-03-30T03:28:00.513563Z","shell.execute_reply.started":"2022-03-30T03:27:59.295138Z","shell.execute_reply":"2022-03-30T03:28:00.512525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:28:00.516011Z","iopub.execute_input":"2022-03-30T03:28:00.516900Z","iopub.status.idle":"2022-03-30T03:28:00.539363Z","shell.execute_reply.started":"2022-03-30T03:28:00.516847Z","shell.execute_reply":"2022-03-30T03:28:00.538671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count Na in %\ndf_sample.isnull().sum()/df_sample.isnull().count()*100","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:28:00.540679Z","iopub.execute_input":"2022-03-30T03:28:00.541825Z","iopub.status.idle":"2022-03-30T03:28:00.593449Z","shell.execute_reply.started":"2022-03-30T03:28:00.541762Z","shell.execute_reply":"2022-03-30T03:28:00.591357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_sample['FN'].unique())\nprint(df_sample['Active'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:28:00.596640Z","iopub.execute_input":"2022-03-30T03:28:00.597306Z","iopub.status.idle":"2022-03-30T03:28:00.605619Z","shell.execute_reply.started":"2022-03-30T03:28:00.597263Z","shell.execute_reply":"2022-03-30T03:28:00.604552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill nan in FN and Active columns with 0\ndf_sample['FN'] = df_sample['FN'].fillna(0)\ndf_sample['Active'] = df_sample['Active'].fillna(0)\n\ndf_sample.isnull().sum()/df_sample.isnull().count()*100","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:28:00.609264Z","iopub.execute_input":"2022-03-30T03:28:00.610159Z","iopub.status.idle":"2022-03-30T03:28:00.665506Z","shell.execute_reply.started":"2022-03-30T03:28:00.610097Z","shell.execute_reply":"2022-03-30T03:28:00.664069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Imputate others with most frequen value \nfrom sklearn.impute import SimpleImputer\n\nimputer = SimpleImputer(strategy='most_frequent')\nimputed_df_sample = pd.DataFrame(imputer.fit_transform(df_sample))\n\nimputed_df_sample.columns = df_sample.columns\n\nprint(imputed_df_sample.isnull().sum()/imputed_df_sample.isnull().count()*100)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:28:00.667400Z","iopub.execute_input":"2022-03-30T03:28:00.667801Z","iopub.status.idle":"2022-03-30T03:28:02.222830Z","shell.execute_reply.started":"2022-03-30T03:28:00.667752Z","shell.execute_reply":"2022-03-30T03:28:02.220713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Object data to category\nfor col in imputed_df_sample.select_dtypes(include=['object']).columns:\n    imputed_df_sample[col] = imputed_df_sample[col].astype('category')\n\nfrom sklearn.preprocessing import OrdinalEncoder\nordinal_encoder = OrdinalEncoder()\n\nobject_cols = imputed_df_sample.select_dtypes(include=['category']).columns\n\n# Encoding categorical features\n# for col in imputed_df_sample.select_dtypes(include=['category']).columns:\n\n\nimputed_df_sample[object_cols] = ordinal_encoder.fit_transform(imputed_df_sample[object_cols])\n    \n    \n# t_dat to datetime objecct\nimputed_df_sample['t_dat'] = pd.to_datetime(imputed_df_sample['t_dat'])\n    \nimputed_df_sample.info()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:28:02.224335Z","iopub.execute_input":"2022-03-30T03:28:02.224611Z","iopub.status.idle":"2022-03-30T03:28:02.613884Z","shell.execute_reply.started":"2022-03-30T03:28:02.224572Z","shell.execute_reply":"2022-03-30T03:28:02.612857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputed_df_sample","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:28:02.615329Z","iopub.execute_input":"2022-03-30T03:28:02.615611Z","iopub.status.idle":"2022-03-30T03:28:02.651099Z","shell.execute_reply.started":"2022-03-30T03:28:02.615577Z","shell.execute_reply":"2022-03-30T03:28:02.650104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop price to prevent data leakage\nimputed_df_sample = imputed_df_sample.drop(['price'], axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot correlation matrix \nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\nplt.figure(figsize=[7,5])\nsns.heatmap(imputed_df_sample.corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:30:15.164212Z","iopub.execute_input":"2022-03-30T03:30:15.165105Z","iopub.status.idle":"2022-03-30T03:30:15.548343Z","shell.execute_reply.started":"2022-03-30T03:30:15.165056Z","shell.execute_reply":"2022-03-30T03:30:15.547065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# select target and features\ny = imputed_df_sample['article_id']\nselected_columns = ['sales_channel_id', 'fashion_news_frequency' , 'postal_code']\n\n# spilt train and test data\nX_train, X_valid, y_train, y_valid = train_test_split(imputed_df_sample[selected_columns], y, test_size = 0.3)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:33:55.300364Z","iopub.execute_input":"2022-03-30T03:33:55.301370Z","iopub.status.idle":"2022-03-30T03:33:55.311518Z","shell.execute_reply.started":"2022-03-30T03:33:55.301321Z","shell.execute_reply":"2022-03-30T03:33:55.310806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create model\nfrom sklearn.ensemble import RandomForestClassifier\n\nmodel = RandomForestClassifier(n_estimators=150, n_jobs=1, max_depth=7)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:33:57.320105Z","iopub.execute_input":"2022-03-30T03:33:57.320627Z","iopub.status.idle":"2022-03-30T03:33:57.325205Z","shell.execute_reply.started":"2022-03-30T03:33:57.320571Z","shell.execute_reply":"2022-03-30T03:33:57.324321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:33:59.059362Z","iopub.execute_input":"2022-03-30T03:33:59.059718Z","iopub.status.idle":"2022-03-30T03:34:33.966839Z","shell.execute_reply.started":"2022-03-30T03:33:59.059684Z","shell.execute_reply":"2022-03-30T03:34:33.965949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting\npredict_labels = model.predict(X_valid)\nprint(predict_labels)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:34:37.148850Z","iopub.execute_input":"2022-03-30T03:34:37.151620Z","iopub.status.idle":"2022-03-30T03:34:58.242793Z","shell.execute_reply.started":"2022-03-30T03:34:37.151562Z","shell.execute_reply":"2022-03-30T03:34:58.241780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#evaluate\nfrom sklearn.metrics import mean_absolute_error\nmean_absolute_error(predict_labels, y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-03-30T03:35:03.718049Z","iopub.execute_input":"2022-03-30T03:35:03.718610Z","iopub.status.idle":"2022-03-30T03:35:03.727610Z","shell.execute_reply.started":"2022-03-30T03:35:03.718558Z","shell.execute_reply":"2022-03-30T03:35:03.726721Z"},"trusted":true},"execution_count":null,"outputs":[]}]}