{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# H&M Recommendation - EDA & Rule Base by Customer Age\n\nThank you for your checking this notebook.\n\nThis is my notebook for \"H&M Personalized Fashion Recommendations\" competition [(Link)](https://www.kaggle.com/c/h-and-m-personalized-fashion-recommendations/overview) to predict purchasing articles based on rule base & customer age.\n\nIf you think this notebook is interesting, please leave your comment or question and I appreciate your upvote as well. :) \n\n<a id='top'></a>\n## Contents\n1. [Import Library & Set Config](#config)\n2. [Load Data](#load)\n3. [EDA of recent popular articles of each ages](#eda)\n4. [Prediction](#pred)\n5. [Submission](#sub)\n6. [Conclution](#conclution)\n7. [Reference](#ref)","metadata":{}},{"cell_type":"markdown","source":"<a id='config'></a>\n\n---\n## 1. Import Library & Set Config\n---\n\n[Back to Contents](#top)","metadata":{}},{"cell_type":"code","source":"# === General ===\nimport sys, warnings, time, os, copy, gc, re, random, pickle, cudf\nwarnings.filterwarnings('ignore')\nfrom IPython.display import display\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\n# pd.set_option('display.max_rows', 50)\n# pd.set_option('display.max_columns', None)\n# pd.set_option(\"display.max_colwidth\", 10000)\nimport seaborn as sns\nsns.set()\nfrom pandas.io.json import json_normalize\nfrom pprint import pprint\nfrom pathlib import Path\nfrom tqdm import tqdm\ntqdm.pandas()\nfrom collections import Counter\nfrom datetime import datetime, timedelta","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:38:21.772707Z","iopub.execute_input":"2022-04-19T16:38:21.773078Z","iopub.status.idle":"2022-04-19T16:38:25.209525Z","shell.execute_reply.started":"2022-04-19T16:38:21.772996Z","shell.execute_reply":"2022-04-19T16:38:25.208731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False\nPATH_INPUT = r'../input/h-and-m-personalized-fashion-recommendations/'","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:38:33.856381Z","iopub.execute_input":"2022-04-19T16:38:33.856849Z","iopub.status.idle":"2022-04-19T16:38:33.86056Z","shell.execute_reply.started":"2022-04-19T16:38:33.856811Z","shell.execute_reply":"2022-04-19T16:38:33.859594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='load'></a>\n\n---\n## 2. Load Data\n---\n\n[Back to Contents](#top)","metadata":{}},{"cell_type":"code","source":"def display_df(df, head=3):\n    print(f'The shape of df is {df.shape}.\\n')\n    display(df.head(head))","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:38:37.524173Z","iopub.execute_input":"2022-04-19T16:38:37.524925Z","iopub.status.idle":"2022-04-19T16:38:37.529837Z","shell.execute_reply.started":"2022-04-19T16:38:37.52488Z","shell.execute_reply":"2022-04-19T16:38:37.528746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfArticles = cudf.read_csv(PATH_INPUT + 'articles.csv', usecols=['article_id', \"product_group_name\", \"perceived_colour_master_name\"])\ndisplay_df(dfArticles, head=3)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:38:41.239578Z","iopub.execute_input":"2022-04-19T16:38:41.24012Z","iopub.status.idle":"2022-04-19T16:38:43.683471Z","shell.execute_reply.started":"2022-04-19T16:38:41.240074Z","shell.execute_reply":"2022-04-19T16:38:43.682817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfCustomers = cudf.read_csv(PATH_INPUT + 'customers.csv', usecols=['customer_id', 'age'])\ndisplay_df(dfCustomers, head=3)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:53:11.74432Z","iopub.execute_input":"2022-04-19T16:53:11.745103Z","iopub.status.idle":"2022-04-19T16:53:11.936496Z","shell.execute_reply.started":"2022-04-19T16:53:11.745055Z","shell.execute_reply":"2022-04-19T16:53:11.935746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We could see age ranging from 16 to 99\nprint(\"max age: \",dfCustomers['age'].max())\nprint(\"min age: \",dfCustomers['age'].min())","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:53:14.565052Z","iopub.execute_input":"2022-04-19T16:53:14.565314Z","iopub.status.idle":"2022-04-19T16:53:14.574984Z","shell.execute_reply.started":"2022-04-19T16:53:14.565286Z","shell.execute_reply":"2022-04-19T16:53:14.574063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let us remove all null values ","metadata":{}},{"cell_type":"code","source":"#Let us split age buckets into 10\ndfCustomers = dfCustomers.to_pandas()\ndfCustomers = dfCustomers[dfCustomers['age'].notna()]\nlistBin = [6,16,26,36,46,56,66,76,86,96,100]\ndfCustomers['age_bins'] = pd.cut(dfCustomers['age'], listBin)\ndisplay_df(dfCustomers, head=3)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:53:16.999667Z","iopub.execute_input":"2022-04-19T16:53:17.000189Z","iopub.status.idle":"2022-04-19T16:53:17.967997Z","shell.execute_reply.started":"2022-04-19T16:53:17.000154Z","shell.execute_reply":"2022-04-19T16:53:17.967355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = dfCustomers[dfCustomers['age_bins'].isnull()].shape[0]\nprint(f'{x} customer_id do not have age information.\\n')","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:53:20.582473Z","iopub.execute_input":"2022-04-19T16:53:20.583227Z","iopub.status.idle":"2022-04-19T16:53:20.589569Z","shell.execute_reply.started":"2022-04-19T16:53:20.583187Z","shell.execute_reply":"2022-04-19T16:53:20.588755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfCustomers.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:53:23.541027Z","iopub.execute_input":"2022-04-19T16:53:23.541575Z","iopub.status.idle":"2022-04-19T16:53:23.68114Z","shell.execute_reply.started":"2022-04-19T16:53:23.541537Z","shell.execute_reply":"2022-04-19T16:53:23.680357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfCustomers[dfCustomers['age_bins'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-04-19T16:53:26.806463Z","iopub.execute_input":"2022-04-19T16:53:26.80717Z","iopub.status.idle":"2022-04-19T16:53:26.81701Z","shell.execute_reply.started":"2022-04-19T16:53:26.807137Z","shell.execute_reply":"2022-04-19T16:53:26.816287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTransactions = cudf.read_csv(PATH_INPUT + 'transactions_train.csv',  \n                               usecols=['t_dat', 'customer_id', 'article_id'],\n                               dtype={'article_id': 'int32', 't_dat': 'string', 'customer_id': 'string'})\ndfTransactions['t_dat'] = cudf.to_datetime(dfTransactions['t_dat'])\n","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:01:58.708552Z","iopub.execute_input":"2022-04-19T17:01:58.708823Z","iopub.status.idle":"2022-04-19T17:02:00.748247Z","shell.execute_reply.started":"2022-04-19T17:01:58.708795Z","shell.execute_reply":"2022-04-19T17:02:00.747538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfTransactions.columns","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:02:04.740163Z","iopub.execute_input":"2022-04-19T17:02:04.740645Z","iopub.status.idle":"2022-04-19T17:02:04.748049Z","shell.execute_reply.started":"2022-04-19T17:02:04.740592Z","shell.execute_reply":"2022-04-19T17:02:04.747381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (dfTransactions.index.min())\nprint (dfTransactions.index.max())","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:02:06.962338Z","iopub.execute_input":"2022-04-19T17:02:06.963045Z","iopub.status.idle":"2022-04-19T17:02:06.974911Z","shell.execute_reply.started":"2022-04-19T17:02:06.963005Z","shell.execute_reply":"2022-04-19T17:02:06.974007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We will take last two months \ndfTransactions.set_index('t_dat', inplace=True)\ndisplay_df(dfTransactions, head=3)\ndfRecent = dfTransactions.loc['2020-08-01' : '2020-09-22']\ndisplay_df(dfRecent, head=3)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:02:10.050515Z","iopub.execute_input":"2022-04-19T17:02:10.051042Z","iopub.status.idle":"2022-04-19T17:02:10.136853Z","shell.execute_reply.started":"2022-04-19T17:02:10.051007Z","shell.execute_reply":"2022-04-19T17:02:10.136156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='eda'></a>\n\n---\n## 3. EDA of recent popular articles of each ages\n\n- Check the latest popular articles in each ages btw. 2020-08-01 and 2020-09-22.\n- Compare that whether is there any difference btw. ages.\n\n---\n\n[Back to Contents](#top)","metadata":{}},{"cell_type":"code","source":"dfRecent = dfRecent.to_pandas()\ndfRecent = dfRecent.merge(dfCustomers[['customer_id', 'age_bins']], on='customer_id', how='inner')\ndisplay_df(dfRecent, head=3)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:02:55.003485Z","iopub.execute_input":"2022-04-19T17:02:55.004047Z","iopub.status.idle":"2022-04-19T17:02:56.5795Z","shell.execute_reply.started":"2022-04-19T17:02:55.004008Z","shell.execute_reply":"2022-04-19T17:02:56.578763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfRecent = dfRecent.groupby(['age_bins', 'article_id']).count().reset_index().rename(columns={'customer_id': 'counts'})\n\n","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:03:54.416635Z","iopub.execute_input":"2022-04-19T17:03:54.417342Z","iopub.status.idle":"2022-04-19T17:03:58.231Z","shell.execute_reply.started":"2022-04-19T17:03:54.417302Z","shell.execute_reply":"2022-04-19T17:03:58.230243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_df(dfRecent, head=10)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:04:58.776366Z","iopub.execute_input":"2022-04-19T17:04:58.777059Z","iopub.status.idle":"2022-04-19T17:04:58.787111Z","shell.execute_reply.started":"2022-04-19T17:04:58.777021Z","shell.execute_reply":"2022-04-19T17:04:58.786371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"listUniBins = dfRecent['age_bins'].unique().tolist()\n\ndict100 = {}\nfor uniBin in listUniBins:\n    dfTemp = dfRecent[dfRecent['age_bins'] == uniBin]\n    dfTemp = dfTemp.sort_values(by='counts', ascending=False)\n    dict100[uniBin] = dfTemp.head(100)['article_id'].values.tolist()\n\ndf100 = pd.DataFrame([dict100]).T.rename(columns={0:'top100'})","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:07:21.220875Z","iopub.execute_input":"2022-04-19T17:07:21.221149Z","iopub.status.idle":"2022-04-19T17:07:21.26446Z","shell.execute_reply.started":"2022-04-19T17:07:21.221119Z","shell.execute_reply":"2022-04-19T17:07:21.263779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index in df100.index:\n    df100[index] = [len(set(df100.at[index, 'top100']) & set(df100.at[x, 'top100']))/100 for x in df100.index]\n\ndf100 = df100.drop(columns='top100')\nplt.figure(figsize=(10, 6))\nsns.heatmap(df100, annot=True, cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:09:30.319298Z","iopub.execute_input":"2022-04-19T17:09:30.319593Z","iopub.status.idle":"2022-04-19T17:09:30.998969Z","shell.execute_reply.started":"2022-04-19T17:09:30.319561Z","shell.execute_reply":"2022-04-19T17:09:30.998295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The most smilar age is (56, 66] & (66, 76], 0.66.\n- The most NOT similar age is (96,100] & (86,96], 0.02.\n","metadata":{}},{"cell_type":"markdown","source":"<a id='pred'></a>\n\n---\n## 4. Prediction\n\n- Predict articles for each age and save the results as csv file separately.\n- Prediction is done by the rule base learned from the notebook [H&M: Faster Trending Products Weekly by Mr. HERVIND PHILIPE](https://www.kaggle.com/code/hervind/h-m-faster-trending-products-weekly/notebook). (Please check and upvote it.)\n\n---\n\n[Back to Contents](#top)","metadata":{}},{"cell_type":"code","source":"N = 12\nlistUniBins = dfCustomers['age_bins'].unique().tolist()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:12:45.33088Z","iopub.execute_input":"2022-04-19T17:12:45.331153Z","iopub.status.idle":"2022-04-19T17:12:45.345736Z","shell.execute_reply.started":"2022-04-19T17:12:45.331123Z","shell.execute_reply":"2022-04-19T17:12:45.344998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for uniBin in listUniBins:\n    df  = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n                            usecols= ['t_dat', 'customer_id', 'article_id'], \n                            dtype={'article_id': 'int32', 't_dat': 'string', 'customer_id': 'string'})\n    if str(uniBin) == 'nan':\n        dfCustomersTemp = dfCustomers[dfCustomers['age_bins'].isnull()]\n    else:\n        dfCustomersTemp = dfCustomers[dfCustomers['age_bins'] == uniBin]\n    \n    dfCustomersTemp = dfCustomersTemp.drop(['age_bins'], axis=1)\n    dfCustomersTemp = cudf.from_pandas(dfCustomersTemp)\n    \n    df = df.merge(dfCustomersTemp[['customer_id', 'age']], on='customer_id', how='inner')\n    print(f'The shape of scope transaction for {uniBin} is {df.shape}. \\n')\n          \n    df ['customer_id'] = df ['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n    df['t_dat'] = cudf.to_datetime(df['t_dat'])\n    last_ts = df['t_dat'].max()\n\n    tmp = df[['t_dat']].copy().to_pandas()\n    tmp['dow'] = tmp['t_dat'].dt.dayofweek\n    tmp['ldbw'] = tmp['t_dat'] - pd.TimedeltaIndex(tmp['dow'] - 1, unit='D')\n    tmp.loc[tmp['dow'] >=2 , 'ldbw'] = tmp.loc[tmp['dow'] >=2 , 'ldbw'] + pd.TimedeltaIndex(np.ones(len(tmp.loc[tmp['dow'] >=2])) * 7, unit='D')\n\n    df['ldbw'] = tmp['ldbw'].values\n    \n    weekly_sales = df.drop('customer_id', axis=1).groupby(['ldbw', 'article_id']).count().reset_index()\n    weekly_sales = weekly_sales.rename(columns={'t_dat': 'count'})\n    \n    df = df.merge(weekly_sales, on=['ldbw', 'article_id'], how = 'left')\n    \n    weekly_sales = weekly_sales.reset_index().set_index('article_id')\n\n    df = df.merge(\n        weekly_sales.loc[weekly_sales['ldbw']==last_ts, ['count']],\n        on='article_id', suffixes=(\"\", \"_targ\"))\n\n    df['count_targ'].fillna(0, inplace=True)\n    del weekly_sales\n    \n    df['quotient'] = df['count_targ'] / df['count']\n    \n    target_sales = df.drop('customer_id', axis=1).groupby('article_id')['quotient'].sum()\n    general_pred = target_sales.nlargest(N).index.to_pandas().tolist()\n    general_pred = ['0' + str(article_id) for article_id in general_pred]\n    general_pred_str =  ' '.join(general_pred)\n    del target_sales\n    \n    purchase_dict = {}\n\n    tmp = df.copy().to_pandas()\n    tmp['x'] = ((last_ts - tmp['t_dat']) / np.timedelta64(1, 'D')).astype(int)\n    tmp['dummy_1'] = 1 \n    tmp['x'] = tmp[[\"x\", \"dummy_1\"]].max(axis=1)\n\n    a, b, c, d = 2.5e4, 1.5e5, 2e-1, 1e3\n    tmp['y'] = a / np.sqrt(tmp['x']) + b * np.exp(-c*tmp['x']) - d\n\n    tmp['dummy_0'] = 0 \n    tmp['y'] = tmp[[\"y\", \"dummy_0\"]].max(axis=1)\n    tmp['value'] = tmp['quotient'] * tmp['y'] \n\n    tmp = tmp.groupby(['customer_id', 'article_id']).agg({'value': 'sum'})\n    tmp = tmp.reset_index()\n\n    tmp = tmp.loc[tmp['value'] > 0]\n    tmp['rank'] = tmp.groupby(\"customer_id\")[\"value\"].rank(\"dense\", ascending=False)\n    tmp = tmp.loc[tmp['rank'] <= 12]\n\n    purchase_df = tmp.sort_values(['customer_id', 'value'], ascending = False).reset_index(drop = True)\n    purchase_df['prediction'] = '0' + purchase_df['article_id'].astype(str) + ' '\n    purchase_df = purchase_df.groupby('customer_id').agg({'prediction': sum}).reset_index()\n    purchase_df['prediction'] = purchase_df['prediction'].str.strip()\n    purchase_df = cudf.DataFrame(purchase_df)\n    \n    sub  = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv',\n                            usecols= ['customer_id'], \n                            dtype={'customer_id': 'string'})\n    \n    numCustomers = sub.shape[0]\n    \n    sub = sub.merge(dfCustomersTemp[['customer_id', 'age']], on='customer_id', how='inner')\n\n    sub['customer_id2'] = sub['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n\n    sub = sub.merge(purchase_df, left_on = 'customer_id2', right_on = 'customer_id', how = 'left',\n                   suffixes = ('', '_ignored'))\n\n    sub = sub.to_pandas()\n    sub['prediction'] = sub['prediction'].fillna(general_pred_str)\n    sub['prediction'] = sub['prediction'] + ' ' +  general_pred_str\n    sub['prediction'] = sub['prediction'].str.strip()\n    sub['prediction'] = sub['prediction'].str[:131]\n    sub = sub[['customer_id', 'prediction']]\n    sub.to_csv(f'submission_' + str(uniBin) + '.csv',index=False)\n    print(f'Saved prediction for {uniBin}. The shape is {sub.shape}. \\n')\n    print('-'*50)\nprint('Finished.\\n')\nprint('='*50)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:13:05.60218Z","iopub.execute_input":"2022-04-19T17:13:05.604809Z","iopub.status.idle":"2022-04-19T17:14:19.236975Z","shell.execute_reply.started":"2022-04-19T17:13:05.604761Z","shell.execute_reply":"2022-04-19T17:14:19.236173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='sub'></a>\n\n---\n## 5. Submission\n\n- Load the saved prediction csv files and concatenate them in one dataframe.\n- Save it as submission.csv.\n\n---\n\n[Back to Contents](#top)","metadata":{}},{"cell_type":"code","source":"for i, uniBin in enumerate(listUniBins):\n    dfTemp  = cudf.read_csv(f'submission_' + str(uniBin) + '.csv')\n    if i == 0:\n        dfSub = dfTemp\n    else:\n        dfSub = cudf.concat([dfSub, dfTemp], axis=0)\n\n#assert dfSub.shape[0] == numCustomers, f'The number of dfSub rows is not correct. {dfSub.shape[0]} vs {numCustomers}.'\n\ndfSub.to_csv(f'submission.csv', index=False)\nprint(f'Saved submission.csv.')","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:15:35.656412Z","iopub.execute_input":"2022-04-19T17:15:35.656966Z","iopub.status.idle":"2022-04-19T17:15:36.838284Z","shell.execute_reply.started":"2022-04-19T17:15:35.65693Z","shell.execute_reply":"2022-04-19T17:15:36.837543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfSub","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:15:45.76123Z","iopub.execute_input":"2022-04-19T17:15:45.761505Z","iopub.status.idle":"2022-04-19T17:15:45.821334Z","shell.execute_reply.started":"2022-04-19T17:15:45.761463Z","shell.execute_reply":"2022-04-19T17:15:45.820529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:30:02.851202Z","iopub.execute_input":"2022-04-19T17:30:02.851486Z","iopub.status.idle":"2022-04-19T17:30:05.058238Z","shell.execute_reply.started":"2022-04-19T17:30:02.851455Z","shell.execute_reply":"2022-04-19T17:30:05.05751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:30:12.935352Z","iopub.execute_input":"2022-04-19T17:30:12.93582Z","iopub.status.idle":"2022-04-19T17:30:12.945616Z","shell.execute_reply.started":"2022-04-19T17:30:12.935781Z","shell.execute_reply":"2022-04-19T17:30:12.944818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfSub.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:30:28.349988Z","iopub.execute_input":"2022-04-19T17:30:28.350555Z","iopub.status.idle":"2022-04-19T17:30:28.371095Z","shell.execute_reply.started":"2022-04-19T17:30:28.350511Z","shell.execute_reply":"2022-04-19T17:30:28.370395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:36:14.234114Z","iopub.execute_input":"2022-04-19T17:36:14.234382Z","iopub.status.idle":"2022-04-19T17:36:14.240242Z","shell.execute_reply.started":"2022-04-19T17:36:14.234352Z","shell.execute_reply":"2022-04-19T17:36:14.239476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfSub = sample_sub.combine_first(dfSub.to_pandas())\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfSub = dfSub.drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:35:56.493703Z","iopub.execute_input":"2022-04-19T17:35:56.494364Z","iopub.status.idle":"2022-04-19T17:35:57.548476Z","shell.execute_reply.started":"2022-04-19T17:35:56.49432Z","shell.execute_reply":"2022-04-19T17:35:57.54773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfSub.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:35:59.496533Z","iopub.execute_input":"2022-04-19T17:35:59.497165Z","iopub.status.idle":"2022-04-19T17:35:59.501739Z","shell.execute_reply.started":"2022-04-19T17:35:59.497128Z","shell.execute_reply":"2022-04-19T17:35:59.501041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfCheck = cudf.read_csv('./submission.csv')\ndisplay_df(dfCheck, head=3)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T17:36:19.858318Z","iopub.execute_input":"2022-04-19T17:36:19.85887Z","iopub.status.idle":"2022-04-19T17:36:20.071586Z","shell.execute_reply.started":"2022-04-19T17:36:19.85883Z","shell.execute_reply":"2022-04-19T17:36:20.070883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='conclution'></a>\n\n---\n\n## 6. Conclution\n\nThank you for your reading through this Notebook!\n\nIf you think this notebook is interesting for you, please do click upvote :)\n\n---\n\n[Back to Contents](#top)","metadata":{}},{"cell_type":"markdown","source":"<a id='ref'></a>\n\n---\n## 7. Reference\n\n1.  [H&M: Faster Trending Products Weekly by Mr. HERVIND PHILIPE](https://www.kaggle.com/code/hervind/h-m-faster-trending-products-weekly/notebook)\n\n---\n\n[Back to Contents](#top)","metadata":{}}]}