{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-02-11T02:55:08.042728Z","iopub.execute_input":"2022-02-11T02:55:08.043872Z","iopub.status.idle":"2022-02-11T02:55:09.145325Z","shell.execute_reply.started":"2022-02-11T02:55:08.043744Z","shell.execute_reply":"2022-02-11T02:55:09.144331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\",nrows=5000)\nprint(articles.columns.tolist())\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\",nrows=5000)\nprint(customers.columns.tolist())\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\",nrows=500)\nprint(transactions.columns.tolist())","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:24:53.388081Z","iopub.execute_input":"2022-02-11T03:24:53.388455Z","iopub.status.idle":"2022-02-11T03:24:53.467491Z","shell.execute_reply.started":"2022-02-11T03:24:53.388415Z","shell.execute_reply":"2022-02-11T03:24:53.466808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merge\ntrans_arti = transactions.merge(articles,how='inner',on=\"article_id\")\ntrans_arti.columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:24:56.33584Z","iopub.execute_input":"2022-02-11T03:24:56.336155Z","iopub.status.idle":"2022-02-11T03:24:56.353419Z","shell.execute_reply.started":"2022-02-11T03:24:56.336123Z","shell.execute_reply":"2022-02-11T03:24:56.352569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trans_arti.head(4)\nprint(\"The shape of the dataset\",trans_arti.shape)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:24:59.756012Z","iopub.execute_input":"2022-02-11T03:24:59.756914Z","iopub.status.idle":"2022-02-11T03:24:59.764775Z","shell.execute_reply.started":"2022-02-11T03:24:59.756833Z","shell.execute_reply":"2022-02-11T03:24:59.763247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n#need to reduce few unimportance Features /columns which can be further used for feature eng. but as here cant due to lack of kaggle RAM\ntrans_arti.head(3)\ndrop_list = [\"index_code\",\"index_name\",\"index_group_no\",\"section_no\",\"detail_desc\",\"prod_name\",\"detail_desc\"\n             ]\ntrans_arti.drop(drop_list,inplace=True,axis=1)\ntrans_arti.head(3)\nprint(\"The shape of the dataset\", trans_arti.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:25:01.765112Z","iopub.execute_input":"2022-02-11T03:25:01.765393Z","iopub.status.idle":"2022-02-11T03:25:01.98207Z","shell.execute_reply.started":"2022-02-11T03:25:01.765365Z","shell.execute_reply":"2022-02-11T03:25:01.979981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merge2\ndf = trans_arti.merge(customers, how=\"inner\", on=\"customer_id\")\ndf.head(3)\nprint(\"The shape of the dataset\", df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:25:04.351118Z","iopub.execute_input":"2022-02-11T03:25:04.351394Z","iopub.status.idle":"2022-02-11T03:25:04.365411Z","shell.execute_reply.started":"2022-02-11T03:25:04.351367Z","shell.execute_reply":"2022-02-11T03:25:04.364516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop the unimportant features\ndrop_col1 =[\"postal_code\"]\ndf.drop(drop_col1,inplace=True,axis=1)\nprint(\"The shape of the dataset\", df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:25:06.855868Z","iopub.execute_input":"2022-02-11T03:25:06.85663Z","iopub.status.idle":"2022-02-11T03:25:06.863122Z","shell.execute_reply.started":"2022-02-11T03:25:06.856594Z","shell.execute_reply":"2022-02-11T03:25:06.862302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking for missing values\ndf.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:25:09.807838Z","iopub.execute_input":"2022-02-11T03:25:09.808782Z","iopub.status.idle":"2022-02-11T03:25:09.817837Z","shell.execute_reply.started":"2022-02-11T03:25:09.808723Z","shell.execute_reply":"2022-02-11T03:25:09.816874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df[[\"Active\",\"FN\"]].dtypes\n#wrong format \ndf[\"Active\"] = df[\"Active\"].astype(object)\ndf[\"Active\"].dtype\ndf[\"FN\"] = df[\"FN\"].astype(object)\n#df[\"Active\"].fillna(df[\"Active\"].mode()[0],inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:25:13.762166Z","iopub.execute_input":"2022-02-11T03:25:13.762632Z","iopub.status.idle":"2022-02-11T03:25:13.769431Z","shell.execute_reply.started":"2022-02-11T03:25:13.762596Z","shell.execute_reply":"2022-02-11T03:25:13.768596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Impute Method 1 with Xgboost\ndef impute_missing_wxgboost(df,conservative = False, n_feats = 10,            \n           fix_string_nans = True, verbose = True,                \n           multiprocessing_load = 3, fill_nans_in_pure_text = True,                    \n           drop_empty_cols = True, drop_nan_cols_with_constant = False,                        \n           ):\n    \n    from verstack import NaNImputer\n    #verstack 2.0.1\n    imputer = NaNImputer(conservative = conservative, n_feats = n_feats,            \n           fix_string_nans = fix_string_nans, verbose = verbose,                \n           multiprocessing_load = multiprocessing_load, fill_nans_in_pure_text = fill_nans_in_pure_text,                    \n           drop_empty_cols = drop_empty_cols, drop_nan_cols_with_constant = drop_nan_cols_with_constant,                      \n           )\n    \n    df_imputed = imputer.impute(df)\n    \n    return df_imputed\n\ndf1 = impute_missing_wxgboost(df)\ndf1.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:25:16.829385Z","iopub.execute_input":"2022-02-11T03:25:16.830284Z","iopub.status.idle":"2022-02-11T03:25:19.218653Z","shell.execute_reply.started":"2022-02-11T03:25:16.830227Z","shell.execute_reply":"2022-02-11T03:25:19.217596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Custom Impute Missing Values\n\n#function to check impute missing values \n#1 checks for datatype\n#2 if character impute mode \n#3 if numberical  check again skewness\n#4 if skewed impute with median else mean\nfrom scipy.stats import skew\n\ndef impute_values(df,skewness_threshold =0.25):\n        for i in df.columns:\n            #temp = df.dtypes\n            if df[i].dtype == object:\n                print(\"Character Variable Detected\")\n                print(\"Missing Value found\",df[i].isna().sum())\n                print(\"Ratio\",df[i].isna().sum()/df.shape[0]*100)\n                df[i].fillna(df[i].mode()[0],inplace=True)\n                print(\"Missing Values After imputation\",df[i].isna().sum())\n            elif df[i].dtype == int or float:\n                s_v = skew(df[i])\n                print(\"Numeric Variable Detected\")\n                if (s_v >=skewness_threshold or s_v <= skewness_threshold):\n                    print(\"Missing Value found\",df[i].isna().sum())\n                    print(\"Skewness detected\")\n                    df[i].fillna(df[i].median(),inplace=True)\n                    print(\"Missing Values After imputation\",df[i].isna().sum())\n                else:\n                    df[i].fillna(df[i].mean(skipna=True),inplace=True)\n                    print(\"Missing Values After imputation\",df[i].isna().sum())\n\n            else:\n                pass\n\n\ndf = impute_values(df)\n            \n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-02-11T03:25:23.013192Z","iopub.execute_input":"2022-02-11T03:25:23.013462Z","iopub.status.idle":"2022-02-11T03:25:23.068559Z","shell.execute_reply.started":"2022-02-11T03:25:23.013435Z","shell.execute_reply":"2022-02-11T03:25:23.06764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T02:56:15.907192Z","iopub.execute_input":"2022-02-11T02:56:15.907486Z","iopub.status.idle":"2022-02-11T02:56:15.917618Z","shell.execute_reply.started":"2022-02-11T02:56:15.907458Z","shell.execute_reply":"2022-02-11T02:56:15.916967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(4)\n#df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-02-10T11:56:50.063096Z","iopub.execute_input":"2022-02-10T11:56:50.063371Z","iopub.status.idle":"2022-02-10T11:56:50.092065Z","shell.execute_reply.started":"2022-02-10T11:56:50.063343Z","shell.execute_reply":"2022-02-10T11:56:50.09109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data is cleaned \n\nto be continued..","metadata":{},"execution_count":null,"outputs":[]}]}