{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Customers Who Bought This Frequently Buy This!\nIn this notebook we will explore which items were frequently purchased together. Using this information, we can predict which items a customer will buy after we observe what they have already bought!","metadata":{"execution":{"iopub.status.busy":"2022-05-29T06:40:30.009139Z","iopub.execute_input":"2022-05-29T06:40:30.00949Z","iopub.status.idle":"2022-05-29T06:40:30.01836Z","shell.execute_reply.started":"2022-05-29T06:40:30.009461Z","shell.execute_reply":"2022-05-29T06:40:30.016676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cudf, gc\nimport cv2, matplotlib.pyplot as plt\nfrom os.path import exists\nprint('RAPIDS version',cudf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T06:40:30.046499Z","iopub.execute_input":"2022-05-29T06:40:30.047135Z","iopub.status.idle":"2022-05-29T06:40:30.055855Z","shell.execute_reply.started":"2022-05-29T06:40:30.047101Z","shell.execute_reply":"2022-05-29T06:40:30.053981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Tranactions","metadata":{}},{"cell_type":"code","source":"# LOAD TRANSACTIONS DATAFRAME\ndf = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\nprint('Transactions shape',df.shape)\ndisplay( df.head() )\n\n# REDUCE MEMORY OF DATAFRAME\ndf = df[['customer_id','article_id']]\ndf.customer_id = df.customer_id.str[-16:].str.hex_to_int().astype('int64')\ndf.article_id = df.article_id.astype('int32')\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T06:47:47.33948Z","iopub.execute_input":"2022-05-29T06:47:47.340421Z","iopub.status.idle":"2022-05-29T06:47:50.649182Z","shell.execute_reply.started":"2022-05-29T06:47:47.340388Z","shell.execute_reply":"2022-05-29T06:47:50.648139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find Items Purchased Together[Products With More Affinity]\nWe will use RAPID cuDF to speed up the dataframe search commands below","metadata":{}},{"cell_type":"code","source":"# FIND ITEMS PURCHASED TOGETHER\nvc = df.article_id.value_counts()\npairs = {}\nfor j,i in enumerate(vc.index.values[1000:1032]):\n    #if j%10==0: print(j,', ',end='')\n    USERS = df.loc[df.article_id==i.item(),'customer_id'].unique()\n    vc2 = df.loc[(df.customer_id.isin(USERS))&(df.article_id!=i.item()),'article_id'].value_counts()\n    pairs[i.item()] = [vc2.index[0], vc2.index[1], vc2.index[2]]","metadata":{"execution":{"iopub.status.busy":"2022-05-29T06:47:55.007328Z","iopub.execute_input":"2022-05-29T06:47:55.007642Z","iopub.status.idle":"2022-05-29T06:48:03.073745Z","shell.execute_reply.started":"2022-05-29T06:47:55.00761Z","shell.execute_reply":"2022-05-29T06:48:03.072731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Display Item Purchased Together\nWhen customers bought the item in the 1st column below, then those customers also bought the items in the 2nd, 3rd, and 4th column too!","metadata":{}},{"cell_type":"code","source":"items = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\nBASE = '../input/h-and-m-personalized-fashion-recommendations/images/'\n\nfor i,(k,v) in enumerate( pairs.items() ):\n    name1 = BASE+'0'+str(k)[:2]+'/0'+str(k)+'.jpg'\n    name2 = BASE+'0'+str(v[0])[:2]+'/0'+str(v[0])+'.jpg'\n    name3 = BASE+'0'+str(v[1])[:2]+'/0'+str(v[1])+'.jpg'\n    name4 = BASE+'0'+str(v[2])[:2]+'/0'+str(v[2])+'.jpg'\n    if exists(name1) & exists(name2) & exists(name3) & exists(name4):\n        plt.figure(figsize=(20,5))\n        img1 = cv2.imread(name1)[:,:,::-1]\n        img2 = cv2.imread(name2)[:,:,::-1]\n        img3 = cv2.imread(name3)[:,:,::-1]\n        img4 = cv2.imread(name4)[:,:,::-1]\n        plt.subplot(1,4,1)\n        plt.title('When customers buy this',size=18)\n        plt.imshow(img1)\n        plt.subplot(1,4,2)\n        plt.title('They buy this',size=18)\n        plt.imshow(img2)\n        plt.subplot(1,4,3)\n        plt.title('They buy this',size=18)\n        plt.imshow(img3)\n        plt.subplot(1,4,4)\n        plt.title('They buy this',size=18)\n        plt.imshow(img4)\n        plt.show()\n    #if i==63: break","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-29T06:40:41.500287Z","iopub.execute_input":"2022-05-29T06:40:41.500634Z","iopub.status.idle":"2022-05-29T06:41:28.636634Z","shell.execute_reply.started":"2022-05-29T06:40:41.500593Z","shell.execute_reply":"2022-05-29T06:41:28.633162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# **BestSellers [Top Sold Products]**","metadata":{"execution":{"iopub.status.busy":"2022-05-29T06:26:12.319251Z","iopub.execute_input":"2022-05-29T06:26:12.319987Z","iopub.status.idle":"2022-05-29T06:26:12.356741Z","shell.execute_reply.started":"2022-05-29T06:26:12.31986Z","shell.execute_reply":"2022-05-29T06:26:12.35556Z"}}},{"cell_type":"code","source":"#imports\nimport pandas as pd\nimport cudf, gc\nimport cv2, matplotlib.pyplot as plt\nfrom os.path import exists\ntransactions = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\narticles = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\n\n\n# Get data\ntop_sold_products = cudf.DataFrame()\ntop_sold_products = transactions[\"article_id\"].value_counts().reset_index().head(100)\ntop_sold_products.columns = [\"article_id\", \"count\"]\nprint(top_sold_products)\n# top_sold_products = cudf.merge(top_sold_products, articles, on=\"article_id\")[[\"article_id\", \"count\", \"prod_name\"]]","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:39:37.542434Z","iopub.execute_input":"2022-05-29T07:39:37.542725Z","iopub.status.idle":"2022-05-29T07:39:40.136419Z","shell.execute_reply.started":"2022-05-29T07:39:37.542692Z","shell.execute_reply":"2022-05-29T07:39:40.135431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_sold_products = cudf.merge(top_sold_products, articles, on=\"article_id\",how='inner')\n# top_sold_products.columns\ntop_sold_products","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:40:06.632298Z","iopub.execute_input":"2022-05-29T07:40:06.632601Z","iopub.status.idle":"2022-05-29T07:40:06.829281Z","shell.execute_reply.started":"2022-05-29T07:40:06.632571Z","shell.execute_reply":"2022-05-29T07:40:06.828398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Getting the Top sold products dictionary\ntop_sold_products = top_sold_products.sort_values('count', ascending=False).reset_index(drop=True)\ntop_sold_products = top_sold_products[['prod_name','count','article_id']]\ntop_sold_products_dict = top_sold_products.to_pandas().to_dict()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:10:58.638158Z","iopub.execute_input":"2022-05-29T07:10:58.638513Z","iopub.status.idle":"2022-05-29T07:10:58.656512Z","shell.execute_reply.started":"2022-05-29T07:10:58.638479Z","shell.execute_reply":"2022-05-29T07:10:58.655497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport os\nos.getcwd()\ntop_sold_products.to_csv(\"/kaggle/working/most_sold_products_final.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:40:48.650689Z","iopub.execute_input":"2022-05-29T07:40:48.651013Z","iopub.status.idle":"2022-05-29T07:40:48.664916Z","shell.execute_reply.started":"2022-05-29T07:40:48.65098Z","shell.execute_reply":"2022-05-29T07:40:48.663865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Trending Categories [Most Sold Categories Recently]","metadata":{}},{"cell_type":"code","source":"#imports\nimport pandas as pd\nimport cudf, gc\nimport cv2, matplotlib.pyplot as plt\nfrom os.path import exists\ntransactions = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\narticles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-29T12:41:14.111057Z","iopub.execute_input":"2022-05-29T12:41:14.111955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#converting the t_dat column into datetime\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\n\n#Getting the time last two months \nstart_date = cudf.to_datetime(\"2020-07-22\")\nend_date = cudf.to_datetime(\"2020-09-22\")\n\n#greater than the start date and smaller than the end date\ntemp = cudf.DataFrame()\nmask = transactions[(transactions['t_dat'] > start_date) & (transactions['t_dat'] <= end_date)]\nmask = mask.reset_index(drop=True)\nmask = mask[['t_dat','article_id']]\ntemp = pd.merge(mask,articles,on='article_id',how='left')\n","metadata":{"execution":{"iopub.status.busy":"2022-05-29T08:42:29.776939Z","iopub.execute_input":"2022-05-29T08:42:29.777217Z","iopub.status.idle":"2022-05-29T08:42:41.837923Z","shell.execute_reply.started":"2022-05-29T08:42:29.777188Z","shell.execute_reply":"2022-05-29T08:42:41.834019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"end_date = pd.to_datetime(transactions['t_dat']).max()\nm = end_date.month\ny = end_date.year\nd = end_date.day\nfor i in range(3):\n    start_dd = str(1)+'-'+str(m)+'-'+str(y)\n    end_dd = str(d)+'-'+str(m)+'-'+str(y)\n    temp = cudf.DataFrame()\n    prod_type = cudf.DataFrame()\n    mask = transactions[(transactions['t_dat'] > start_dd) & (transactions['t_dat'] <= end_dd)]\n    mask = mask.reset_index(drop=True)\n    mask = mask[['t_dat','article_id']]\n    temp = cudf.merge(mask,articles,on='article_id',how='left')\n    prod_type = temp[\"product_type_name\"].value_counts().reset_index().head(100)\n    prod_type.columns = [\"product_type_name\", \"count\"]\n    prod_type['date'] =  end_dd\n    prod_type.to_csv(\"/kaggle/working/trending_categories_\"+end_dd+\".csv\", index=False)\n    m-=1\n    ","metadata":{"execution":{"iopub.status.busy":"2022-05-29T09:31:15.075088Z","iopub.execute_input":"2022-05-29T09:31:15.075343Z","iopub.status.idle":"2022-05-29T09:32:28.567944Z","shell.execute_reply.started":"2022-05-29T09:31:15.075314Z","shell.execute_reply":"2022-05-29T09:32:28.56673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions['t_dat'].min(),transactions['t_dat'].max()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T08:46:33.303712Z","iopub.execute_input":"2022-05-29T08:46:33.303985Z","iopub.status.idle":"2022-05-29T08:46:33.488012Z","shell.execute_reply.started":"2022-05-29T08:46:33.303957Z","shell.execute_reply":"2022-05-29T08:46:33.487118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prod_type = temp[\"product_type_name\"].value_counts().reset_index().head(100)\nprod_type.columns = [\"product_type_name\", \"count\"]\nprod_type.to_csv(\"/kaggle/working/trending_categories.csv\", index=False)\nprod_type","metadata":{"execution":{"iopub.status.busy":"2022-05-29T09:00:26.63331Z","iopub.execute_input":"2022-05-29T09:00:26.633565Z","iopub.status.idle":"2022-05-29T09:00:26.646442Z","shell.execute_reply.started":"2022-05-29T09:00:26.633536Z","shell.execute_reply":"2022-05-29T09:00:26.645572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Trending Colors","metadata":{}},{"cell_type":"code","source":"#imports\nimport pandas as pd\nimport cudf, gc\nimport cv2, matplotlib.pyplot as plt\nfrom os.path import exists\ntransactions = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\narticles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"end_date = pd.to_datetime(transactions['t_dat']).max()\nm = end_date.month\ny = end_date.year\nd = end_date.day\nfor i in range(3):\n    start_dd = str(1)+'-'+str(m)+'-'+str(y)\n    end_dd = str(d)+'-'+str(m)+'-'+str(y)\n    temp = cudf.DataFrame()\n    prod_type = cudf.DataFrame()\n    mask = transactions[(transactions['t_dat'] > start_dd) & (transactions['t_dat'] <= end_dd)]\n    mask = mask.reset_index(drop=True)\n    mask = mask[['t_dat','article_id']]\n    temp = pd.merge(mask,articles,on='article_id',how='left')\n    prod_type = temp[\"colour_group_name\"].value_counts().reset_index().head(100)\n    prod_type.columns = [\"colour_group_name\", \"count\"]\n    prod_type['date'] =  end_dd\n    prod_type.to_csv(\"/kaggle/working/trending_colortypes_\"+end_dd+\".csv\", index=False)\n    m-=1","metadata":{"execution":{"iopub.status.busy":"2022-05-29T10:16:12.975187Z","iopub.execute_input":"2022-05-29T10:16:12.975456Z","iopub.status.idle":"2022-05-29T10:17:24.432023Z","shell.execute_reply.started":"2022-05-29T10:16:12.975426Z","shell.execute_reply":"2022-05-29T10:17:24.431337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Weekly Trending Products","metadata":{}},{"cell_type":"code","source":"#imports\nimport pandas as pd\nimport cudf, gc\nimport cv2, matplotlib.pyplot as plt\nfrom os.path import exists\ntransactions = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\narticles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\nend_date = cudf.to_datetime(transactions['t_dat']).max()\nfor i in range(4):\n    if (end_date.day-7) % 30 == 0:\n        new_day = 1\n    else:\n        new_day = (end_date.day-7) % 30\n    month = end_date.month\n    year = end_date.year\n    start_date = cudf.to_datetime(str(new_day)+\"-\"+str(month)+\"-\"+str(year))\n    temp = cudf.DataFrame()\n    prod_type = cudf.DataFrame()\n    mask = transactions[(transactions['t_dat'] > start_date) & (transactions['t_dat'] <= end_date)]\n    mask = mask.reset_index(drop=True)\n    mask = mask[['t_dat','article_id']]\n    temp = cudf.merge(mask,articles,on='article_id',how='left')\n    prod_type = temp[\"prod_name\"].value_counts().reset_index().head(100)\n    prod_type.columns = [\"prod_name\", \"count\"]\n    prod_type['date'] =  end_date\n    prod_type.to_csv(\"/kaggle/working/weekly_trending_products_\"+str(end_date)+\".csv\", index=False)\n    end_date = start_date\n    ","metadata":{"execution":{"iopub.status.busy":"2022-05-29T13:03:50.187287Z","iopub.execute_input":"2022-05-29T13:03:50.187519Z","iopub.status.idle":"2022-05-29T13:04:18.522097Z","shell.execute_reply.started":"2022-05-29T13:03:50.187487Z","shell.execute_reply":"2022-05-29T13:04:18.52142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}