{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport scipy.stats as stats\n\nimport vaex\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.preprocessing import FunctionTransformer\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.compose       import ColumnTransformer\n\nimport warnings\n\n\n# These are the 4 csv files' paths:\n# 1. /kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\n# 2. /kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\n# 3. /kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\n# 4. /kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv\n\n# And this is one of the images path:\n# 1. /kaggle/input/h-and-m-personalized-fashion-recommendations/images/057/0570177001.jpg\n\n\n# hide unwanted warning comming from pandas dataframe operations\npd.options.mode.chained_assignment = None\n# display all the columns, (don't hide some columns while viewing the dataframe)\npd.options.display.max_columns     = None\n# hide other unwanted warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-24T03:05:12.496771Z","iopub.execute_input":"2022-02-24T03:05:12.497132Z","iopub.status.idle":"2022-02-24T03:05:15.765977Z","shell.execute_reply.started":"2022-02-24T03:05:12.497035Z","shell.execute_reply":"2022-02-24T03:05:15.764599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1. EDA</span>\n\n## <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.1 \"article\" csv file</span>\n\n**<span style=\"color:#023e8a\">Details of the `articles` csv file:</span>**\n- **<span style=\"color:#B016E4\">article_id</span>**<span style=\"color:#023e8a;\">: **A unique identifier of every article.**</span>\n- **<span style=\"color:#B016E4\">product_code, prod_name</span>**<span style=\"color:#023e8a;\">: **A unique identifier of every product and its name (not the same).</span>**  \n- **<span style=\"color:#B016E4\">product_type, product_type_name</span>**<span style=\"color:#023e8a;\">: **The group of product_code and its name</span>**  \n- **<span style=\"color:#B016E4\">graphical_appearance_no, graphical_appearance_name</span>**<span style=\"color:#023e8a;\">: **The group of graphics and its name</span>**  \n- **<span style=\"color:#B016E4\">colour_group_code, colour_group_name</span>**<span style=\"color:#023e8a;\">: **The group of color and its name</span>**  \n- **<span style=\"color:#B016E4\">graphical_appearance_no, graphical_appearance_name</span>**<span style=\"color:#023e8a;\">: **The group of graphics and its name</span>**  \n- **<span style=\"color:#B016E4\">perceived_colour_value_id, perceived_colour_value_name, perceived_colour_master_id, perceived_colour_master_name</span>**<span style=\"color:#023e8a;\">: **The added color info</span>**  \n- **<span style=\"color:#B016E4\">department_no, department_name:</span>**<span style=\"color:#023e8a;\">: **A unique identifier of every dep and its name</span>**  \n- **<span style=\"color:#B016E4\">index_code, index_name:</span>**<span style=\"color:#023e8a;\">: **A unique identifier of every index and its name</span>**  \n- **<span style=\"color:#B016E4\">index_group_no, index_group_name:</span>**<span style=\"color:#023e8a;\">: **A group of indeces and its name</span>**  \n- **<span style=\"color:#B016E4\">section_no, section_name:</span>**<span style=\"color:#023e8a;\">: **A unique identifier of every section and its name</span>**  \n- **<span style=\"color:#B016E4\">garment_group_no, garment_group_name:</span>**<span style=\"color:#023e8a;\">: **A unique identifier of every garment and its name</span>**  \n- **<span style=\"color:#B016E4\">detail_desc:</span>**<span style=\"color:#023e8a;\">: **Short description</span>**  ","metadata":{}},{"cell_type":"code","source":"df_articles = vaex.from_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\")\nprint(f\"Shape of the articles dataset: {df_articles.shape}\")\ndf_articles.sample(3)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:06:53.337759Z","iopub.execute_input":"2022-02-24T03:06:53.338028Z","iopub.status.idle":"2022-02-24T03:06:54.598074Z","shell.execute_reply.started":"2022-02-24T03:06:53.337999Z","shell.execute_reply":"2022-02-24T03:06:54.597277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2 \"transaction\" csv file</span>\n\n**<span style=\"color:#023e8a\">Details of the `transactions` csv file:<span>**\n- **<span style=\"color:#B016E4\">t_dat</span>**<span style=\"color:#023e8a;\">: **Transaction date.**</span>\n- **<span style=\"color:#B016E4\">customer_id</span>**<span style=\"color:#023e8a;\">: **A unique identifier of every customer.</span>**  \n- **<span style=\"color:#B016E4\">article_id</span>**<span style=\"color:#023e8a;\">: **A unique identifier of every article (from `articles` dataframe) cuntomer bought</span>**  \n- **<span style=\"color:#B016E4\">price</span>**<span style=\"color:#023e8a;\">: **The customer spend how much money</span>**\n- **<span style=\"color:#B016E4\">sales_channel_id</span>**<span style=\"color:#023e8a;\">: **1 or 2</span>**","metadata":{}},{"cell_type":"code","source":"df_transaction = vaex.from_csv(\n    \"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\",\n    dtype={\"sales_channel_id\": \"int8\", \"article_id\": \"int32\", \"price\": \"float32\"} \n)\nprint(f\"Shape of transaction dataset: {df_transaction.shape}\")\ndf_transaction.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:07:27.761677Z","iopub.execute_input":"2022-02-24T03:07:27.761960Z","iopub.status.idle":"2022-02-24T03:08:58.758445Z","shell.execute_reply.started":"2022-02-24T03:07:27.761930Z","shell.execute_reply":"2022-02-24T03:08:58.757458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.1 Check that is there any missing values</span>","metadata":{}},{"cell_type":"code","source":"def vaex_is_null(df):\n    count_na = []\n    for col in df.column_names:\n        count_na.append(df[col].isna().sum().item())\n    return pd.Series(data=count_na, index=df.column_names).sort_values(ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:18:34.834389Z","iopub.execute_input":"2022-02-24T03:18:34.834847Z","iopub.status.idle":"2022-02-24T03:18:34.840885Z","shell.execute_reply.started":"2022-02-24T03:18:34.834798Z","shell.execute_reply":"2022-02-24T03:18:34.839892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vaex_is_null(df_transaction)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:19:06.808820Z","iopub.execute_input":"2022-02-24T03:19:06.809123Z","iopub.status.idle":"2022-02-24T03:19:11.173946Z","shell.execute_reply.started":"2022-02-24T03:19:06.809091Z","shell.execute_reply":"2022-02-24T03:19:11.173136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.2 Statistical Analysis</span>\n    \n**<span style=\"color:#023e8a;\">For statistical analysis, we can only use the \"price\" column.</span>**","metadata":{}},{"cell_type":"code","source":"# short description of \"price\" column by removing the scientific notation\ndf_transaction.describe()","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:31:30.915424Z","iopub.execute_input":"2022-02-24T03:31:30.915856Z","iopub.status.idle":"2022-02-24T03:31:36.316050Z","shell.execute_reply.started":"2022-02-24T03:31:30.915825Z","shell.execute_reply":"2022-02-24T03:31:36.315487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.2 Distribution of \"price\" column</span>\n\n**<span style=\"color:#023e8a;\">According to Freedman-Diaconis:</span>**\n\n<span style=\"color:#023e8a; font-size:2em\">\n$$h = 2\\frac{IQR}{\\sqrt[3]{n}}$$\n</span>\n\n**<span style=\"color:#023e8a;\">And then, the number of bins (k) in a histogram should be:</span>**\n\n<span style=\"color:#023e8a; font-size:2em\">\n    $$k = \\frac{max(x) - min(x)}{h}$$\n</span>","metadata":{}},{"cell_type":"code","source":"def calculate_number_of_bins(data: pd.DataFrame) -> int:\n    # calculate the 75th percentile\n    q3  = np.quantile(data.evaluate(), 0.75)\n    # calculate the 25th percentile\n    q1  = np.quantile(data.evaluate(), 0.25)\n    # calcutate IQR\n    iqr = q3 - q1\n    # calculate total number of records\n    n   = data.shape[0]\n    # calcute the Freedman-Diaconis\n    h   = 2 * (iqr/(np.cbrt(n)))\n    # calculate the number of bins\n    k   = (df_transaction[\"price\"].max() - df_transaction[\"price\"].min())/h\n    return int(k)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:46:25.332197Z","iopub.execute_input":"2022-02-24T03:46:25.332528Z","iopub.status.idle":"2022-02-24T03:46:25.340323Z","shell.execute_reply.started":"2022-02-24T03:46:25.332495Z","shell.execute_reply":"2022-02-24T03:46:25.339481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see the distribution of the \"price\" column\nplt.figure(figsize=(16, 9))\nsns.set_style(\"darkgrid\")\nsns.distplot(df_transaction[\"price\"].evaluate(), hist=False, color=\"#16E437\", bins=calculate_number_of_bins(df_transaction[\"price\"]))\nplt.xlabel(\"Price\",   fontsize=16)\nplt.ylabel(\"Density\", fontsize=16)\nplt.title(\"Distribution of the \\\"price\\\" column\", fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:46:30.505480Z","iopub.execute_input":"2022-02-24T03:46:30.505791Z","iopub.status.idle":"2022-02-24T03:48:19.727387Z","shell.execute_reply.started":"2022-02-24T03:46:30.505762Z","shell.execute_reply":"2022-02-24T03:48:19.726305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**<span style=\"color:#023e8a;\">From the distribution, we can see that the \"price\" column is right skewed data. It is not a unimodal distribution. There are more than one peaks in the distribution. Now lets see how much differ the distribution from the normal distribution using `QQ plot`.</span>**","metadata":{}},{"cell_type":"markdown","source":"### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.3 QQ Plot of \"price\" column</span>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 9))\nx = stats.probplot(df_transaction[\"price\"].evaluate(), plot=plt)\nplt.xlabel(\"Theoritical quantities\", fontsize=16)\nplt.ylabel(\"Ordered Values\", fontsize=16)\nplt.title(\"QQ Plot of \\\"price\\\" column\", fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:48:38.310036Z","iopub.execute_input":"2022-02-24T03:48:38.310347Z","iopub.status.idle":"2022-02-24T03:49:45.045263Z","shell.execute_reply.started":"2022-02-24T03:48:38.310315Z","shell.execute_reply":"2022-02-24T03:49:45.044156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.4 BoxPlot of \"price\" column</span>","metadata":{}},{"cell_type":"code","source":"# lets see the boxplot\nplt.figure(figsize=(16, 9))\nsns.boxplot(x=df_transaction[\"price\"].evaluate(), color=\"#B616E4\")\nplt.title('BoxPlot of \"price\" column', fontsize=16)\nplt.xlabel(\"Price\", fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:51:27.040384Z","iopub.execute_input":"2022-02-24T03:51:27.040732Z","iopub.status.idle":"2022-02-24T03:51:31.398905Z","shell.execute_reply.started":"2022-02-24T03:51:27.040701Z","shell.execute_reply":"2022-02-24T03:51:31.397663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**<span style=\"color:#023e8a;\">Because of highly right skewed data (\"price\" column), so many outliers are detected. Because of the non-normal distribution, we can remove the outliers using the IQR method. First we have to transform this to a normal distribution.</span>**\n\n**<span style=\"color:#023e8a;\">For skewed distribution, usually these below 3 methods are used to convert to normal distribution:</span>**\n- **<span style=\"color:#023e8a;\">Log Transform</span>**\n- **<span style=\"color:#023e8a;\">Square Transform</span>**\n- **<span style=\"color:#023e8a;\">Box-Cox Transform</span>**","metadata":{}},{"cell_type":"markdown","source":"### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.5 Log transformation of \"price\" column</span>","metadata":{}},{"cell_type":"code","source":"# log transformation\nlog_transformer = FunctionTransformer(func=np.log1p)\n\ndf_transaction[\"price_log_transform\"] = log_transformer.fit_transform(df_transaction[\"price\"])","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:53:59.205072Z","iopub.execute_input":"2022-02-24T03:53:59.205864Z","iopub.status.idle":"2022-02-24T03:53:59.212100Z","shell.execute_reply.started":"2022-02-24T03:53:59.205811Z","shell.execute_reply":"2022-02-24T03:53:59.211262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see the result of log transformation\nplt.figure(figsize=(16, 9))\nsns.set_style(\"darkgrid\")\nsns.distplot(df_transaction[\"price_log_transform\"].evaluate(), hist=False, color=\"#16E437\", bins=calculate_number_of_bins(df_transaction[\"price_log_transform\"]))\nplt.xlabel(\"Price (log transformed)\", fontsize=16)\nplt.ylabel(\"Density\", fontsize=16)\nplt.title(\"Distribution of the log transformed \\\"price\\\" column\", fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:54:15.054832Z","iopub.execute_input":"2022-02-24T03:54:15.055391Z","iopub.status.idle":"2022-02-24T03:56:06.665498Z","shell.execute_reply.started":"2022-02-24T03:54:15.055357Z","shell.execute_reply":"2022-02-24T03:56:06.664364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.6 Square transformation of \"price\" column</span>","metadata":{}},{"cell_type":"code","source":"# square transformation\nsquare_transformer = FunctionTransformer(func=np.square)\n\ndf_transaction[\"price_square_transform\"] = square_transformer.fit_transform(df_transaction[\"price\"])","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:57:31.261339Z","iopub.execute_input":"2022-02-24T03:57:31.262104Z","iopub.status.idle":"2022-02-24T03:57:31.266925Z","shell.execute_reply.started":"2022-02-24T03:57:31.262063Z","shell.execute_reply":"2022-02-24T03:57:31.266260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see the result of square transformation\nplt.figure(figsize=(16, 9))\nsns.set_style(\"darkgrid\")\nsns.distplot(df_transaction[\"price_square_transform\"].evaluate(), hist=False, color=\"#16E437\")\nplt.xlabel(\"Price (square transformed)\", fontsize=16)\nplt.ylabel(\"Density\", fontsize=16)\nplt.title(\"Distribution of the square transformed \\\"price\\\" column\", fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:57:41.509851Z","iopub.execute_input":"2022-02-24T03:57:41.510410Z","iopub.status.idle":"2022-02-24T03:59:27.119144Z","shell.execute_reply.started":"2022-02-24T03:57:41.510372Z","shell.execute_reply":"2022-02-24T03:59:27.118187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.7 Box-Cox transformation of \"price\" column</span>","metadata":{}},{"cell_type":"code","source":"# box-cox transformation\nbox_cox_transformer = PowerTransformer(method=\"box-cox\")\n\ndf_transaction[\"price_box_cox_transform\"] = box_cox_transformer.fit_transform(df_transaction[\"price\"].to_numpy().reshape(df_transaction[\"price\"].shape[0], 1)) ","metadata":{"execution":{"iopub.status.busy":"2022-02-24T03:59:34.533224Z","iopub.execute_input":"2022-02-24T03:59:34.533527Z","iopub.status.idle":"2022-02-24T03:59:58.914776Z","shell.execute_reply.started":"2022-02-24T03:59:34.533495Z","shell.execute_reply":"2022-02-24T03:59:58.914011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see the result of Box-Cox transformation\nplt.figure(figsize=(16, 9))\nsns.set_style(\"darkgrid\")\nsns.distplot(df_transaction[\"price_box_cox_transform\"].evaluate(), hist=False, color=\"#16E437\")\nplt.xlabel(\"Price (box-cox transformed)\", fontsize=16)\nplt.ylabel(\"Density\", fontsize=16)\nplt.title(\"Distribution of the \\\"price\\\" column\", fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-24T04:00:15.535058Z","iopub.execute_input":"2022-02-24T04:00:15.535747Z","iopub.status.idle":"2022-02-24T04:02:10.481285Z","shell.execute_reply.started":"2022-02-24T04:00:15.535704Z","shell.execute_reply":"2022-02-24T04:02:10.480495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**<span style=\"color:#023e8a;\">From the results of 3 transformations, we can see that Box-Cox transformation perform well than other 2 transformations.</span>**","metadata":{}},{"cell_type":"markdown","source":"### <span style=\"color:white; background-color:#B616E4; padding:5px; border-radius:7px\">1.2.8 Top 30 customers by number of transactions</span>","metadata":{}},{"cell_type":"markdown","source":"**<span style=\"color:#023e8a;\">First I will apply \"value_counts\" method of the pandas DataFrame on the \"customer_id\" column. It will return the number of transactions in descending order. For easy usage, I have converted this into pandas DataFrame. For better visualization purpose, I change the x-axis values from unique long customer_id value to index value.</span>**","metadata":{}},{"cell_type":"code","source":"# calculate the value counts of \"customer_id\"\ndf_transactions_value_counts                = pd.DataFrame(df_transaction[\"customer_id\"].value_counts())\ndf_transactions_value_counts[\"counts\"]      = df_transactions_value_counts.iloc[:, 0]\ndf_transactions_value_counts[\"customer_id\"] = df_transactions_value_counts.index\ndf_transactions_value_counts.reset_index(drop=True, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-24T04:10:49.144392Z","iopub.execute_input":"2022-02-24T04:10:49.144687Z","iopub.status.idle":"2022-02-24T04:10:58.195913Z","shell.execute_reply.started":"2022-02-24T04:10:49.144658Z","shell.execute_reply":"2022-02-24T04:10:58.195109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now visualise the top 15 customers by number of transactions\nhave_to_display = 30 # define how many customers I want to plot\nplt.figure(figsize=(16, 9))\nsns.barplot(\n    x=df_transactions_value_counts[\"customer_id\"].iloc[:have_to_display],\n    y=df_transactions_value_counts[\"counts\"].iloc[:have_to_display]\n)\nplt.title(\"Top 15 customers by number of transactions\", fontsize=16)\nplt.xlabel(\"Customer ID\", fontsize=16)\nplt.ylabel(\"Total transactions\", fontsize=16)\nplt.xticks(ticks = list(range(have_to_display)), labels=list(range(have_to_display)), rotation=45, fontsize=16)\nplt.yticks(fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-24T04:11:05.082304Z","iopub.execute_input":"2022-02-24T04:11:05.082872Z","iopub.status.idle":"2022-02-24T04:11:05.574351Z","shell.execute_reply.started":"2022-02-24T04:11:05.082824Z","shell.execute_reply":"2022-02-24T04:11:05.573297Z"},"trusted":true},"execution_count":null,"outputs":[]}]}