{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n##### 1) Recommending Top 12 Articles\n##### 2) Recommend items that are bought together with previous purchases","metadata":{}},{"cell_type":"markdown","source":"# If you find this notebook useful or interesting, please, support with an upvote 😊","metadata":{}},{"cell_type":"code","source":"import sys\nimport pandas as pd\nimport numpy as np\nimport scipy.sparse as sparse\nfrom scipy.sparse.linalg import spsolve\nimport random\nfrom sklearn import metrics\nfrom sklearn.preprocessing import MinMaxScaler\nimport implicit\nimport plotly.express as px\n\n#Importing the necessary libraries\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom mpl_toolkits.mplot3d import Axes3D\n%matplotlib inline\n\n\nfrom collections import Counter\nfrom PIL import Image\nfrom pathlib import Path","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-22T03:21:02.203351Z","iopub.execute_input":"2022-02-22T03:21:02.204239Z","iopub.status.idle":"2022-02-22T03:21:05.06232Z","shell.execute_reply.started":"2022-02-22T03:21:02.20412Z","shell.execute_reply":"2022-02-22T03:21:05.060842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{"execution":{"iopub.status.busy":"2022-02-21T20:24:18.556897Z","iopub.execute_input":"2022-02-21T20:24:18.557443Z","iopub.status.idle":"2022-02-21T20:24:18.562716Z","shell.execute_reply.started":"2022-02-21T20:24:18.557406Z","shell.execute_reply":"2022-02-21T20:24:18.561962Z"}}},{"cell_type":"code","source":"%%time\npath = Path(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/\")\n\narticles_df = pd.read_csv(path / \"articles.csv\", dtype = {'article_id': str})\ncust_df = pd.read_csv(path / \"customers.csv\", dtype = {'customer_id': str})\ntrans_df = pd.read_csv(path / \"transactions_train.csv\", dtype = {'article_id': str,'customer_id': str})\ntrans_df[\"t_dat\"] = pd.to_datetime(trans_df[\"t_dat\"])\n# trans_df = trans_df[[\"t_dat\", \"article_id\"]]\nmonthly_df = trans_df.query(\"'2020-9-1' <= t_dat\")\nweekly_df = trans_df.query(\"'2020-9-16' <= t_dat\")\n","metadata":{"execution":{"iopub.status.busy":"2022-02-22T02:41:46.067474Z","iopub.execute_input":"2022-02-22T02:41:46.067748Z","iopub.status.idle":"2022-02-22T02:43:10.127269Z","shell.execute_reply.started":"2022-02-22T02:41:46.067713Z","shell.execute_reply":"2022-02-22T02:43:10.12648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.info()\narticles_df.head()\narticles_df = articles_df[['article_id', 'product_type_name','product_group_name','colour_group_code']]","metadata":{"execution":{"iopub.status.busy":"2022-02-22T02:43:10.128464Z","iopub.execute_input":"2022-02-22T02:43:10.13206Z","iopub.status.idle":"2022-02-22T02:43:10.335164Z","shell.execute_reply.started":"2022-02-22T02:43:10.131987Z","shell.execute_reply":"2022-02-22T02:43:10.334195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust_df.info()\ncust_df = cust_df[['customer_id', 'club_member_status','fashion_news_frequency','age']]\n\n# counting unique customer\nn_cust = len(pd.unique(cust_df['customer_id']))\nprint(\"No.of.unique values :\",n_cust)","metadata":{"execution":{"iopub.status.busy":"2022-02-22T02:43:10.336722Z","iopub.execute_input":"2022-02-22T02:43:10.337814Z","iopub.status.idle":"2022-02-22T02:43:11.827443Z","shell.execute_reply.started":"2022-02-22T02:43:10.337754Z","shell.execute_reply":"2022-02-22T02:43:11.82599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfg = cust_df[['age','fashion_news_frequency','customer_id']]\ndfg = dfg.groupby(['age','fashion_news_frequency']).count().reset_index()\ndfg.rename(columns = {\"customer_id\": \"count\"}, inplace=True)\ndfg\nfig = px.bar(dfg, x=\"age\", y=\"count\",color='fashion_news_frequency'\n#               ,markers=True\n              ,color_discrete_sequence=px.colors.diverging.PRGn\n             ,template = \"plotly_white\"\n             ) \nfig.update_layout(\n    title=\"Number of customer by age\"\n    ,xaxis_title=\"Age\"\n    ,yaxis_title=\"Count\"\n    ,legend_title_text='fashion_news_frequency'\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-22T02:43:11.82902Z","iopub.execute_input":"2022-02-22T02:43:11.829315Z","iopub.status.idle":"2022-02-22T02:43:12.46Z","shell.execute_reply.started":"2022-02-22T02:43:11.829278Z","shell.execute_reply":"2022-02-22T02:43:12.458776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trans_df.info()\n\n# Format date\ntrans_df['t_dat'] = pd.to_datetime(trans_df['t_dat'])\ntrans_df['YYYY_MM'] = trans_df['t_dat'].dt.year.astype(str) + '_' + trans_df['t_dat'].dt.month.astype(str)\ntrans_df['year'] = trans_df['t_dat'].dt.year\ntrans_df['month'] = trans_df['t_dat'].dt.month\n\n# Printing minimum and the maximum date from dataset.\nprint(trans_df['t_dat'].min())\nprint(trans_df['t_dat'].max())","metadata":{"execution":{"iopub.status.busy":"2022-02-22T02:43:12.461441Z","iopub.execute_input":"2022-02-22T02:43:12.461669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Join the dataset - Left Join (Excluse those do not have transaction)\ndf = pd.merge(trans_df, cust_df, on='customer_id', how='left')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# counting unique customer\nn = len(pd.unique(df['customer_id']))\nprint(\"No.of.unique customer that have transaction in transactions_train.csv :\",n)\n\nn_cust_notintran = n_cust - n\nprint(\"No.of.customer that have no transaction in transactions_train.csv :\",n_cust_notintran)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Join the dataset - Left Join (Excluse those do not have transaction)\ndf = pd.merge(df, articles_df, on='article_id', how='left')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_df = df[['t_dat', 'customer_id','article_id']]\ncount_df = count_df.groupby(['t_dat', 'customer_id']).size().rename('quantity').reset_index()\ncount_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of unique customers: {count_df.customer_id.nunique()}')\nprint(f'Number of unique items: {df.article_id.nunique()}')\n\nprint(f'Average purchase quantity per interaction: {int(count_df.quantity.mean())}')\nprint(f'Minimum purchase quantity per interaction: {count_df.quantity.min()}')\nprint(f'Maximum purchase quantity per interaction: {count_df.quantity.max()}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n### Find the Monthly Top 12 Articles \nI would recommend the latest monthly Top 10 items to the customer who does not have transaction(that I can not learn)\nidea from https://www.kaggle.com/negoto/best-selling-items-catalog-like-eda-of-articles","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom collections import Counter\nfrom PIL import Image\nfrom pathlib import Path\n\n\ndef show_images(article_ids, cols=1, rows=-1):\n    if isinstance(article_ids, int) or isinstance(article_ids, str):\n        article_ids = [article_ids]\n    article_count = len(article_ids)\n    if rows < 0: rows = (article_count // cols) + 1\n    plt.figure(figsize=(3 + 3.5 * cols, 3 + 5 * rows))\n    for i in range(article_count):\n        article_id = (\"0\" + str(article_ids[i]))[-10:]\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n        plt.title(article_id)\n        try:\n            image = Image.open(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/{article_id[:3]}/{article_id}.jpg\")\n            plt.imshow(image)\n        except:\n            pass\n\n\nsales_counts = Counter(trans_df.article_id)\nfor i in range(len(articles_df)):\n    articles_df.at[i, \"sales_count\"] = sales_counts[articles_df.at[i, \"article_id\"]]\n\nmonthly_sales_counts = Counter(monthly_df.article_id)\nfor i in range(len(articles_df)):\n    articles_df.at[i, \"monthly_sales_count\"] = monthly_sales_counts[articles_df.at[i, \"article_id\"]]\n    \nweekly_sales_counts = Counter(weekly_df.article_id)\nfor i in range(len(articles_df)):\n    articles_df.at[i, \"weekly_sales_count\"] = weekly_sales_counts[articles_df.at[i, \"article_id\"]]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df = articles_df.sort_values(by=\"monthly_sales_count\", ascending=False)\ntemp = articles_df.article_id[:12]\nshow_images(list(temp), 6)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## work in progress.\n\n## Please do upvote if you like it.Thanks","metadata":{}}]}