{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Prepare data","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nimport numpy as np\nimport pandas as pd\nimport datetime\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-08T21:36:08.962135Z","iopub.execute_input":"2022-05-08T21:36:08.964446Z","iopub.status.idle":"2022-05-08T21:36:08.99152Z","shell.execute_reply.started":"2022-05-08T21:36:08.964313Z","shell.execute_reply":"2022-05-08T21:36:08.990733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load transaction_train.csv data set\ntransactions = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ntransactions.to_parquet('transactions.parquet')\ntransactions_parquet = pd.read_parquet('./transactions.parquet')\ndel transactions","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:36:16.802109Z","iopub.execute_input":"2022-05-08T21:36:16.802764Z","iopub.status.idle":"2022-05-08T21:37:52.2936Z","shell.execute_reply.started":"2022-05-08T21:36:16.802729Z","shell.execute_reply":"2022-05-08T21:37:52.292835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Memory usage reduction\ntransactions_parquet['article_id'] = transactions_parquet['article_id'].astype('int32')\n# transactions_parquet['article_id'] = '0' + transactions_parquet.article_id.astype('str')\n#transactions_parquet['customer_id'] =\\\n#    transactions_parquet['customer_id'].apply(lambda x: int(x[-16:],16) ).astype('int64')\n# sub = cudf.read_csv('sample_submission.csv')[['customer_id']]\n# sub['customer_id_2'] =\\\n#    sub['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n# sub = sub.merge(PREDS_DF.rename({'customer_id':'customer_id_2'},axis=1),\\\n#    on='customer_id_2', how='left').fillna('')\n# del sub['customer_id_2']\n# sub.to_csv('submission.csv',index=False)\ntransactions_parquet['price'] = transactions_parquet['price'].astype('float32')\ntransactions_parquet['sales_channel_id'] = transactions_parquet['sales_channel_id'].astype('int8')","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:38:10.33746Z","iopub.execute_input":"2022-05-08T21:38:10.337917Z","iopub.status.idle":"2022-05-08T21:38:10.562937Z","shell.execute_reply.started":"2022-05-08T21:38:10.337883Z","shell.execute_reply":"2022-05-08T21:38:10.562106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load articles.csv data set\narticles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\narticles.to_parquet('articles.parquet')\narticles_parquet = pd.read_parquet('./articles.parquet')\ndel articles","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:38:18.574501Z","iopub.execute_input":"2022-05-08T21:38:18.57482Z","iopub.status.idle":"2022-05-08T21:38:20.107421Z","shell.execute_reply.started":"2022-05-08T21:38:18.574785Z","shell.execute_reply":"2022-05-08T21:38:20.106526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_parquet.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:38:27.00509Z","iopub.execute_input":"2022-05-08T21:38:27.00542Z","iopub.status.idle":"2022-05-08T21:38:27.040956Z","shell.execute_reply.started":"2022-05-08T21:38:27.005387Z","shell.execute_reply":"2022-05-08T21:38:27.04007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load customers.csv data set\ncustomers = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')\ncustomers.to_parquet('customers.parquet')\ncustomers_parquet = pd.read_parquet('./customers.parquet')\ndel customers\n# Memory usage reduction\n#customers_parquet['customer_id'] =\\\n#    customers_parquet['customer_id'].apply(lambda x: int(x[-16:],16) ).astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:38:57.807997Z","iopub.execute_input":"2022-05-08T21:38:57.808782Z","iopub.status.idle":"2022-05-08T21:39:05.54817Z","shell.execute_reply.started":"2022-05-08T21:38:57.808727Z","shell.execute_reply":"2022-05-08T21:39:05.547267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_parquet.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:39:10.703189Z","iopub.execute_input":"2022-05-08T21:39:10.703498Z","iopub.status.idle":"2022-05-08T21:39:10.719702Z","shell.execute_reply.started":"2022-05-08T21:39:10.703467Z","shell.execute_reply":"2022-05-08T21:39:10.718614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load sample_submission.csv data set\nsub = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\nsub.to_parquet('sub.parquet')\nsub_parquet = pd.read_parquet('./sub.parquet')\ndel sub","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:39:12.788538Z","iopub.execute_input":"2022-05-08T21:39:12.788849Z","iopub.status.idle":"2022-05-08T21:39:20.058362Z","shell.execute_reply.started":"2022-05-08T21:39:12.788803Z","shell.execute_reply":"2022-05-08T21:39:20.057592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_parquet.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:39:25.673109Z","iopub.execute_input":"2022-05-08T21:39:25.673415Z","iopub.status.idle":"2022-05-08T21:39:25.683475Z","shell.execute_reply.started":"2022-05-08T21:39:25.673384Z","shell.execute_reply":"2022-05-08T21:39:25.682575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert t_dat to the datetime format\ntransactions_parquet[\"t_dat\"] = pd.to_datetime(transactions_parquet[\"t_dat\"])","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:39:28.329226Z","iopub.execute_input":"2022-05-08T21:39:28.329928Z","iopub.status.idle":"2022-05-08T21:39:34.576879Z","shell.execute_reply.started":"2022-05-08T21:39:28.329881Z","shell.execute_reply":"2022-05-08T21:39:34.576024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_parquet.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:39:37.711102Z","iopub.execute_input":"2022-05-08T21:39:37.711423Z","iopub.status.idle":"2022-05-08T21:39:37.723711Z","shell.execute_reply.started":"2022-05-08T21:39:37.711387Z","shell.execute_reply":"2022-05-08T21:39:37.722642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Make smaller dataframes to work on.","metadata":{}},{"cell_type":"code","source":"transactions_6w = transactions_parquet[transactions_parquet['t_dat'] >= '2020-08-12'].copy()\nprint(len(transactions_6w)) # 1565245","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:39:43.837713Z","iopub.execute_input":"2022-05-08T21:39:43.83804Z","iopub.status.idle":"2022-05-08T21:39:44.116987Z","shell.execute_reply.started":"2022-05-08T21:39:43.838006Z","shell.execute_reply":"2022-05-08T21:39:44.116004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_1w = transactions_parquet[transactions_parquet['t_dat'] >= '2020-09-16'].copy()\nprint(len(transactions_1w)) # 240311","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:39:51.995495Z","iopub.execute_input":"2022-05-08T21:39:51.996218Z","iopub.status.idle":"2022-05-08T21:39:52.121709Z","shell.execute_reply.started":"2022-05-08T21:39:51.996169Z","shell.execute_reply":"2022-05-08T21:39:52.120821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# For cold starters:\nWe recommend the 12 bestsellers for each age group from 2020-08-12 to 2020-09-22.","metadata":{}},{"cell_type":"code","source":"# Find the list of cold starters\nunique_customers_df = transactions_parquet.drop_duplicates(keep='first', subset='customer_id')\nnum_cold_starters = len(customers_parquet) - len(unique_customers_df)\nprint(num_cold_starters) # 9699\nsimple_customers_df = customers_parquet[['customer_id', 'age']].copy()\nunique_df = simple_customers_df.merge(unique_customers_df, how='left', on='customer_id')\ncold_starters_df = unique_df.loc[unique_df['article_id'].isna()]\ncold_starters_df = cold_starters_df[['customer_id', 'age']]\nprint(cold_starters_df.isna().values.any()) # true\ncold_starters_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:39:58.241826Z","iopub.execute_input":"2022-05-08T21:39:58.242419Z","iopub.status.idle":"2022-05-08T21:40:11.228639Z","shell.execute_reply.started":"2022-05-08T21:39:58.242361Z","shell.execute_reply":"2022-05-08T21:40:11.227778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Segment cold starters by ages, and then recommend each age group by\n# its corresponding list of 12 bestsellers over the last 6 weeks\ntransactions_6w = pd.merge(transactions_6w, customers_parquet, on='customer_id', how='left')\nsimple_transactions_6w = transactions_6w[['customer_id', 'article_id', 'age']]\nsimple_transactions_6w = simple_transactions_6w.groupby(['age', 'article_id']).customer_id.count()\\\n    .reset_index(name='count').sort_values(['age', 'count'], ascending=[True, False])\nitems_for_cold_starters = simple_transactions_6w.groupby('age').head(12).copy()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:40:15.624948Z","iopub.execute_input":"2022-05-08T21:40:15.625283Z","iopub.status.idle":"2022-05-08T21:40:18.576822Z","shell.execute_reply.started":"2022-05-08T21:40:15.625246Z","shell.execute_reply":"2022-05-08T21:40:18.575803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(items_for_cold_starters.isna().values.any()) # false\ndel items_for_cold_starters['count']\nitems_for_cold_starters.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:40:24.225813Z","iopub.execute_input":"2022-05-08T21:40:24.226612Z","iopub.status.idle":"2022-05-08T21:40:24.244192Z","shell.execute_reply.started":"2022-05-08T21:40:24.226565Z","shell.execute_reply":"2022-05-08T21:40:24.24364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can clearly see that there are some age groups with fewer than 12 recommendations, so we fill the remaining with the bestsellers from previous week and select the top 12 recommendations. So does for those with no age information provided. Remove duplicates before concatenating two lists into one string.","metadata":{}},{"cell_type":"code","source":"top12_1w = transactions_1w.groupby('article_id').size().reset_index(name='count')\\\n    .sort_values('count', ascending=False).head(12).copy()\n#top12_1w = ' 0' + ' 0'.join(transactions_1w.article_id.value_counts().index.astype('str')[:12])\n#print(\"Last week's top 12 popular items:\")\n#print(top12_1w)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-08T21:40:29.454081Z","iopub.execute_input":"2022-05-08T21:40:29.454906Z","iopub.status.idle":"2022-05-08T21:40:29.476081Z","shell.execute_reply.started":"2022-05-08T21:40:29.454868Z","shell.execute_reply":"2022-05-08T21:40:29.475333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill the predictions dataframe with last week's 12 bestsellers\n# and select the top 12 recommendations\ntop12_1w.article_id = top12_1w.article_id.astype('str')\nitems_for_cold_starters.article_id = items_for_cold_starters.article_id.astype('str')\n#items_for_cold_starters.article_id = ' 0' + items_for_cold_starters.article_id.astype('str')\nsegments = items_for_cold_starters.copy().reset_index(drop=True)\nsegments = segments.groupby('age')['article_id'].apply(list).reset_index(name='article_id')\npred = []\nfor idx in segments.article_id:\n    pred.append(list(set(idx + top12_1w.article_id.tolist()))[:12])\nsegments = pd.DataFrame({'age':segments.age,'list':pred})\nsegments['list'] = segments['list'].transform(' 0'.join).reset_index(drop=True)\nsegments.list = segments.list.str.strip()\nsegments.list = segments.list.str.zfill(131)\nprint(len(segments))\nsegments\n#segments['article_id'] = segments.groupby('age', as_index=False).article_id.astype('str') + top12_1w.article_id.astype('str').drop_duplicates().transform(' '.join).reset_index(drop=True)\n#segments['article_id'] = segments.groupby('age', as_index=False)['article_id'].transform(' '.join).reset_index(drop=True)\n#segments.rename({'article_id':'list'}, axis=1, inplace=True)\n#segments.drop_duplicates().reset_index(drop=True)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-08T21:40:32.953967Z","iopub.execute_input":"2022-05-08T21:40:32.954294Z","iopub.status.idle":"2022-05-08T21:40:32.987847Z","shell.execute_reply.started":"2022-05-08T21:40:32.954256Z","shell.execute_reply":"2022-05-08T21:40:32.986876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(segments.isna().values.any()) # false","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:40:37.778469Z","iopub.execute_input":"2022-05-08T21:40:37.778753Z","iopub.status.idle":"2022-05-08T21:40:37.783793Z","shell.execute_reply.started":"2022-05-08T21:40:37.778723Z","shell.execute_reply":"2022-05-08T21:40:37.783114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Construct the cold_starter_predictions dataframe\ncold_starter_predictions = cold_starters_df.merge(segments, on='age', how='left').fillna('')\ndel cold_starter_predictions['age']\ncold_starter_predictions = cold_starter_predictions.drop_duplicates().reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:40:39.853194Z","iopub.execute_input":"2022-05-08T21:40:39.853524Z","iopub.status.idle":"2022-05-08T21:40:39.88449Z","shell.execute_reply.started":"2022-05-08T21:40:39.853484Z","shell.execute_reply":"2022-05-08T21:40:39.883517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(cold_starter_predictions.isna().values.any()) # false\ncold_starter_predictions","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:41:00.975883Z","iopub.execute_input":"2022-05-08T21:41:00.97652Z","iopub.status.idle":"2022-05-08T21:41:00.996055Z","shell.execute_reply.started":"2022-05-08T21:41:00.97648Z","shell.execute_reply":"2022-05-08T21:41:00.995086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_0 = sub_parquet[['customer_id']].merge(cold_starter_predictions, on='customer_id', how='left')\n#sub_0['list'] = sub_0['list'].apply(lambda d: d if isinstance(d, list) else [''])\nsub_0.list.fillna('', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T02:09:45.358039Z","iopub.execute_input":"2022-05-09T02:09:45.358569Z","iopub.status.idle":"2022-05-09T02:09:46.780413Z","shell.execute_reply.started":"2022-05-09T02:09:45.358505Z","shell.execute_reply":"2022-05-09T02:09:46.779709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sub_0.isna().values.any()) # false\nsub_0","metadata":{"execution":{"iopub.status.busy":"2022-05-09T07:03:05.99652Z","iopub.execute_input":"2022-05-09T07:03:05.996808Z","iopub.status.idle":"2022-05-09T07:03:06.302744Z","shell.execute_reply.started":"2022-05-09T07:03:05.996778Z","shell.execute_reply":"2022-05-09T07:03:06.301986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# For existing customers, we adopt 4 strategies:\n1. First recommend 1 item - the latest purchase of each customer -\n   to account for return or repeated purchase\n2. **Half-User-Based Collaborative Filtering**: Recommend 3 items that are most often purchased together\n3. Third recommend 3 bestsellers with discount\n4. **Model-Based Collaborative Filtering**: Lastly we implement collaborative filtering algorithm to recommend 5 items that are similar to the customer's purchasing history\n\nIf duplicates occur, replace with items in the lower ranking of (4) to make\nsure we recommend 12 items to every customer (the evaluation metric - map@k12\nassures us that this is actually an optimal strategy)\n\nDue to runtime and memory constraints, we will only work on the transactions\ndata from 2019-08-07 to 2020-09-22.\nFor customers without purchase behavior after 2019-08-07, we assume that they\nwill not purchase anything from 2020-09-23 to 2020-09-29 as well. ","metadata":{}},{"cell_type":"code","source":"transactions_1y = transactions_parquet[transactions_parquet['t_dat'] >= '2019-08-07'].copy()\nprint(len(transactions_1y))","metadata":{"execution":{"iopub.status.busy":"2022-05-08T21:41:41.888924Z","iopub.execute_input":"2022-05-08T21:41:41.889287Z","iopub.status.idle":"2022-05-08T21:41:43.830796Z","shell.execute_reply.started":"2022-05-08T21:41:41.889249Z","shell.execute_reply":"2022-05-08T21:41:43.830072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. H&M has a 30-day return policy, which means customers can still return their purchases as late as on 2020-08-31","metadata":{}},{"cell_type":"code","source":"transactions_1m = transactions_1y[transactions_1y['t_dat'] >= '2020-08-31'].copy()\ntransactions_1m = transactions_1m.sort_values(['customer_id','t_dat'], ascending=False)\ntransactions_1m = transactions_1m[['t_dat', 'customer_id', 'article_id']]\ntransactions_1m = transactions_1m.drop_duplicates(subset=['customer_id'])\nprint(len(transactions_1m)) # 196319\npredictions_1 = transactions_1m[['customer_id', 'article_id']].reset_index(drop=True)\npredictions_1['article_id'] = predictions_1['article_id'].astype(str)\npredictions_1","metadata":{"execution":{"iopub.status.busy":"2022-05-09T04:57:29.108246Z","iopub.execute_input":"2022-05-09T04:57:29.108979Z","iopub.status.idle":"2022-05-09T04:57:30.942516Z","shell.execute_reply.started":"2022-05-09T04:57:29.108933Z","shell.execute_reply":"2022-05-09T04:57:30.941597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(pd.unique(predictions_1['customer_id']))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T04:57:40.540108Z","iopub.execute_input":"2022-05-09T04:57:40.541145Z","iopub.status.idle":"2022-05-09T04:57:40.6913Z","shell.execute_reply.started":"2022-05-09T04:57:40.541084Z","shell.execute_reply":"2022-05-09T04:57:40.69065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The predictions_1 dataframe shows that we will recommend 1 previously purchased item to 267779 customers.","metadata":{}},{"cell_type":"code","source":"sub_1 = sub_parquet[['customer_id']].merge(predictions_1, on='customer_id', how='left')\n#sub_1['article_id'] = sub_1['article_id'].apply(lambda d: d if isinstance(d, list) else [''])\nsub_1.article_id.fillna('', inplace=True)\n#sub = sub[['customer_id']]\n#sub['prediction'] = sub_0['list'] + sub_1['article_id']","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:25:22.455762Z","iopub.execute_input":"2022-05-09T05:25:22.456067Z","iopub.status.idle":"2022-05-09T05:25:23.607274Z","shell.execute_reply.started":"2022-05-09T05:25:22.456031Z","shell.execute_reply":"2022-05-09T05:25:23.606383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_1.count()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:25:32.389145Z","iopub.execute_input":"2022-05-09T05:25:32.389681Z","iopub.status.idle":"2022-05-09T05:25:32.700344Z","shell.execute_reply.started":"2022-05-09T05:25:32.389646Z","shell.execute_reply":"2022-05-09T05:25:32.699458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_1.article_id = sub_1.article_id.str.strip()\nsub_1.article_id = sub_1.article_id.str.zfill(10)\n#sub.prediction = sub.prediction.str[:131]\nsub_1.article_id = sub_1.article_id.apply(lambda d: d if d!='0000000000' else '')\nsub_1","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:25:35.478955Z","iopub.execute_input":"2022-05-09T05:25:35.479622Z","iopub.status.idle":"2022-05-09T05:25:37.377039Z","shell.execute_reply.started":"2022-05-09T05:25:35.479571Z","shell.execute_reply":"2022-05-09T05:25:37.376187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.1 Half-User-Based Collaborative Filtering: Recommend 3 items that are most often purchased together","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-05-08T22:45:05.903733Z","iopub.execute_input":"2022-05-08T22:45:05.904494Z","iopub.status.idle":"2022-05-08T22:45:05.908916Z","shell.execute_reply.started":"2022-05-08T22:45:05.904436Z","shell.execute_reply":"2022-05-08T22:45:05.908025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND ITEMS PURCHASED TOGETHER\nvcc = transactions_6w[['customer_id','article_id']].copy()\nvc = vcc.article_id.value_counts()\nitems = []\narticles = []\nfor i in enumerate(tqdm(vc.index.values)):\n    articles.append(i[1])\n    USERS = transactions_6w.loc[transactions_6w.article_id==i[1],'customer_id'].unique()\n    vc2 = transactions_6w.loc[(transactions_6w.customer_id.isin(USERS))&(transactions_6w.article_id!=i[1]),'article_id'].value_counts()\n    if len(vc2) >= 3:\n        items.append([str(vc2.index[0]), str(vc2.index[1]), str(vc2.index[2])])\n    elif len(vc2) == 2:\n        items.append([str(vc2.index[0]), str(vc2.index[1])])\n    elif len(vc2) == 1:\n        items.append([str(vc2.index[0])])\n    else:\n        items.append([''])","metadata":{"execution":{"iopub.status.busy":"2022-05-08T22:45:13.261853Z","iopub.execute_input":"2022-05-08T22:45:13.262159Z","iopub.status.idle":"2022-05-09T00:34:30.545485Z","shell.execute_reply.started":"2022-05-08T22:45:13.262123Z","shell.execute_reply":"2022-05-09T00:34:30.54479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_2 = pd.DataFrame({'article_id':articles,'list':items})\npredictions_2 = transactions_parquet.merge(predictions_2, on='article_id', how='left')\npredictions_2 = predictions_2.sort_values(['customer_id', 't_dat'], ascending=[False, False])\npredictions_2 = predictions_2.drop_duplicates(subset=['customer_id'])\npredictions_2 = predictions_2[['customer_id', 'list']].reset_index(drop=True)\npredictions_2","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:25:43.54718Z","iopub.execute_input":"2022-05-09T05:25:43.547714Z","iopub.status.idle":"2022-05-09T05:26:29.887672Z","shell.execute_reply.started":"2022-05-09T05:25:43.547673Z","shell.execute_reply":"2022-05-09T05:26:29.886638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Quick check: 1362281+9699 = 1371980.","metadata":{}},{"cell_type":"code","source":"predictions_2['list'] = predictions_2['list'].apply(lambda d: d if isinstance(d, list) else [''])\npredictions_2['list'] = predictions_2['list'].apply(lambda x: ' 0'.join(map(str, x)))\npredictions_2","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:26:55.705523Z","iopub.execute_input":"2022-05-09T05:26:55.705839Z","iopub.status.idle":"2022-05-09T05:26:58.403179Z","shell.execute_reply.started":"2022-05-09T05:26:55.705804Z","shell.execute_reply":"2022-05-09T05:26:58.402344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_2.list = predictions_2.list.str.strip()\npredictions_2.list = ' 0' + predictions_2.list\npredictions_2.list = predictions_2.list.apply(lambda d: d if d!=' 0' else '')\npredictions_2 = sub_parquet[['customer_id']].merge(predictions_2, on='customer_id', how='left')\n#sub_1['article_id'] = sub_1['article_id'].apply(lambda d: d if isinstance(d, list) else [''])\npredictions_2.list.fillna('', inplace=True)\nsub_1['article_id'] = sub_1['article_id'] + predictions_2['list']\nsub_1.article_id = sub_1.article_id.str.strip()\nsub_1","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:27:16.433584Z","iopub.execute_input":"2022-05-09T05:27:16.434221Z","iopub.status.idle":"2022-05-09T05:27:20.625347Z","shell.execute_reply.started":"2022-05-09T05:27:16.434168Z","shell.execute_reply":"2022-05-09T05:27:20.624387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.2 Model-Based Collaborative Filtering: Singular Value Decomposition Model","metadata":{}},{"cell_type":"code","source":"!pip install git+https://github.com/mayukh18/reco -q","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:29:43.062977Z","iopub.execute_input":"2022-05-09T05:29:43.063302Z","iopub.status.idle":"2022-05-09T05:30:21.803059Z","shell.execute_reply.started":"2022-05-09T05:29:43.063267Z","shell.execute_reply":"2022-05-09T05:30:21.801998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from reco.recommender import FunkSVD\nfrom reco.metrics import rmse\nfrom collections import Counter","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:30:55.345693Z","iopub.execute_input":"2022-05-09T05:30:55.34603Z","iopub.status.idle":"2022-05-09T05:31:01.782126Z","shell.execute_reply.started":"2022-05-09T05:30:55.345993Z","shell.execute_reply":"2022-05-09T05:31:01.781386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We first train an SVD model using transactions data from 2020-08-12 to 2020-09-15.\nWe will leave the last week of training as validation data.","metadata":{}},{"cell_type":"code","source":"transactions_5w = transactions_parquet[(transactions_parquet['t_dat'] >= '2020-08-12') & (transactions_parquet['t_dat'] <= '2020-09-15')].copy()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:31:06.816689Z","iopub.execute_input":"2022-05-09T05:31:06.817076Z","iopub.status.idle":"2022-05-09T05:31:07.385731Z","shell.execute_reply.started":"2022-05-09T05:31:06.817031Z","shell.execute_reply":"2022-05-09T05:31:07.384654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = transactions_5w.copy()\npositive_items_per_user = train.groupby(['customer_id'])['article_id'].apply(list)\ntrain['pop_factor'] = train['t_dat'].apply(lambda x: 1/(datetime.datetime(2020,9,16) - x).days**2)\npopular_items_group = train.groupby(['article_id'])['pop_factor'].sum()\n\ntrain['feedback'] = 1\ntrain = train.groupby(['customer_id', 'article_id']).sum().reset_index()\ntrain['feedback'] = train.apply(lambda row: row['feedback']/popular_items_group[row['article_id']], axis=1)\ntrain['feedback'] = train['feedback'].apply(lambda x: 5.0 if x>5.0 else x)\ntrain.drop(['price', 'sales_channel_id'], axis=1, inplace=True)\ntrain = train.sample(frac=1).reset_index(drop=True)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:31:10.514635Z","iopub.execute_input":"2022-05-09T05:31:10.515476Z","iopub.status.idle":"2022-05-09T05:32:17.174735Z","shell.execute_reply.started":"2022-05-09T05:31:10.515428Z","shell.execute_reply":"2022-05-09T05:32:17.1733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = FunkSVD(k=8, learning_rate=0.005, regularizer = .01, iterations = 200, method = 'stochastic', bias=True)\nf.fit(X=train, formatizer={'user':'customer_id', 'item':'article_id', 'value':'feedback'},verbose=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:32:45.027024Z","iopub.execute_input":"2022-05-09T05:32:45.027384Z","iopub.status.idle":"2022-05-09T05:34:14.767805Z","shell.execute_reply.started":"2022-05-09T05:32:45.027347Z","shell.execute_reply":"2022-05-09T05:34:14.766671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we validate our model using transactions data from last week:","metadata":{}},{"cell_type":"code","source":"validation = transactions_1w.copy()\npositive_items_val = validation.groupby(['customer_id'])['article_id'].apply(list)\nval_users = positive_items_val.keys()\nval_items = []\n\nfor i,user in tqdm(enumerate(val_users)):\n    val_items.append(positive_items_val[user])\n    \nprint(\"Total users in validation:\", len(val_users))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:34:26.158118Z","iopub.execute_input":"2022-05-09T05:34:26.158657Z","iopub.status.idle":"2022-05-09T05:34:29.031727Z","shell.execute_reply.started":"2022-05-09T05:34:26.158613Z","shell.execute_reply":"2022-05-09T05:34:29.030189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define the evaluation metric map12.","metadata":{}},{"cell_type":"code","source":"def apk(actual, predicted, k=12):\n    if len(predicted)>k:\n        predicted = predicted[:k]\n\n    score = 0.0\n    num_hits = 0.0\n\n    for i,p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            num_hits += 1.0\n            score += num_hits / (i+1.0)\n\n    if not actual:\n        return 0.0\n\n    return score / min(len(actual), k)\n\ndef mapk(actual, predicted, k=12):\n    return np.mean([apk(a,p,k) for a,p in zip(actual, predicted)])","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:34:32.426671Z","iopub.execute_input":"2022-05-09T05:34:32.42696Z","iopub.status.idle":"2022-05-09T05:34:32.436119Z","shell.execute_reply.started":"2022-05-09T05:34:32.426925Z","shell.execute_reply":"2022-05-09T05:34:32.434592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Make validations:","metadata":{}},{"cell_type":"code","source":"_, popular_items = zip(*sorted(zip(popular_items_group, popular_items_group.keys()))[::-1])\npopular_items = list(popular_items)\nuserindexes = {f.users[i]:i for i in range(len(f.users))}\n\noutputs = []\ncnt = 0\n\nfor user in tqdm(val_users):\n    user_output = []\n    if user in positive_items_per_user:\n        most_common_items_of_user = {k:v for k, v in Counter(positive_items_per_user[user]).most_common()}\n        user_index = userindexes[user]\n        new_order = {}\n        for k in list(most_common_items_of_user.keys())[:20]:\n            try:\n                itemindex = f.items.index(k)\n                pred_value = np.dot(f.userfeatures[user_index], f.itemfeatures[itemindex].T) + f.item_bias[0, itemindex]\n            except:\n                pred_value = most_common_items_of_user[k]\n            new_order[k] = pred_value\n        user_output += [k for k, v in sorted(new_order.items(), key=lambda item: item[1])][:12]\n    \n    user_output += list(popular_items[:12 - len(user_output)])\n    outputs.append(user_output)\n    \nprint(\"MAP Score on Validation set:\", mapk(val_items, outputs))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:34:38.854224Z","iopub.execute_input":"2022-05-09T05:34:38.854672Z","iopub.status.idle":"2022-05-09T05:36:18.602945Z","shell.execute_reply.started":"2022-05-09T05:34:38.854618Z","shell.execute_reply":"2022-05-09T05:36:18.601656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After validating our model, we repeat the the process and train the model on the whole transactions_6w.csv dataframe for submission.","metadata":{}},{"cell_type":"code","source":"train = transactions_6w.copy()\npositive_items_per_user = train.groupby(['customer_id'])['article_id'].apply(list)\ntrain['pop_factor'] = train['t_dat'].apply(lambda x: 1/(datetime.datetime(2020,9,23) - x).days**2)\npopular_items_group = train.groupby(['article_id'])['pop_factor'].sum()\n\ntrain['feedback'] = 1\ntrain = train.groupby(['customer_id', 'article_id']).sum().reset_index()\ntrain['feedback'] = train.apply(lambda row: row['feedback']/popular_items_group[row['article_id']], axis=1)\ntrain['feedback'] = train['feedback'].apply(lambda x: 5.0 if x>5.0 else x)\ntrain.drop(['price', 'sales_channel_id'], axis=1, inplace=True)\ntrain = train.sample(frac=1).reset_index(drop=True)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:36:49.501406Z","iopub.execute_input":"2022-05-09T05:36:49.501688Z","iopub.status.idle":"2022-05-09T05:38:22.058189Z","shell.execute_reply.started":"2022-05-09T05:36:49.501658Z","shell.execute_reply":"2022-05-09T05:38:22.057516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = FunkSVD(k=8, learning_rate=0.005, regularizer = .01, iterations = 200, method = 'stochastic', bias=True)\nf.fit(X=train, formatizer={'user':'customer_id', 'item':'article_id', 'value':'feedback'},verbose=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T05:39:41.48885Z","iopub.execute_input":"2022-05-09T05:39:41.48914Z","iopub.status.idle":"2022-05-09T05:41:43.28295Z","shell.execute_reply.started":"2022-05-09T05:39:41.489109Z","shell.execute_reply":"2022-05-09T05:41:43.282094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, popular_items = zip(*sorted(zip(popular_items_group, popular_items_group.keys()))[::-1])\npopular_items = list(popular_items)\n\nuserindexes = {f.users[i]:i for i in range(len(f.users))}\npositive_items_per_user = train.groupby(['customer_id'])['article_id'].apply(list)\n\noutputs = []\ncnt = 0\n\nfor user in tqdm(sub_parquet['customer_id']):\n    user_output = []\n    if user in positive_items_per_user.keys():\n        most_common_items_of_user = {k:v for k, v in Counter(positive_items_per_user[user]).most_common()}\n        \n        user_index = userindexes[user]\n        new_order = {}\n        for k in list(most_common_items_of_user.keys())[:20]:\n            try:\n                itemindex = f.items.index(k)\n                pred_value = np.dot(f.userfeatures[user_index], f.itemfeatures[itemindex].T) + f.item_bias[0, itemindex]\n            except:\n                pred_value = most_common_items_of_user[k]\n            new_order[k] = pred_value\n        user_output += [k for k, v in sorted(new_order.items(), key=lambda item: item[1])][:12]\n    \n    user_output += list(popular_items[:12 - len(user_output)])\n    outputs.append(user_output)\n    \nusers = []\nfor user in tqdm(sub_parquet['customer_id']):\n    users.append(user)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:10:06.57525Z","iopub.execute_input":"2022-05-09T06:10:06.576072Z","iopub.status.idle":"2022-05-09T06:18:58.854339Z","shell.execute_reply.started":"2022-05-09T06:10:06.576035Z","shell.execute_reply":"2022-05-09T06:18:58.853121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_4 = pd.DataFrame({'customer_id':users,'list':outputs})\nprediction_4['list'] = prediction_4['list'].apply(lambda x: ' 0'.join(map(str, x)))#.reset_index(drop=True)\nprediction_4.list = prediction_4.list.str.strip()\nprediction_4.list = prediction_4.list.str.zfill(131)\nprediction_4 = sub_parquet[['customer_id']].merge(prediction_4, on='customer_id', how='left')","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:24:13.683455Z","iopub.execute_input":"2022-05-09T06:24:13.684184Z","iopub.status.idle":"2022-05-09T06:24:22.493346Z","shell.execute_reply.started":"2022-05-09T06:24:13.684149Z","shell.execute_reply":"2022-05-09T06:24:22.492611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(prediction_4.isna().values.any()) # False\nprediction_4","metadata":{"execution":{"iopub.status.busy":"2022-05-09T07:17:58.630584Z","iopub.execute_input":"2022-05-09T07:17:58.630893Z","iopub.status.idle":"2022-05-09T07:17:58.954836Z","shell.execute_reply.started":"2022-05-09T07:17:58.630859Z","shell.execute_reply":"2022-05-09T07:17:58.954273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Run this in the end!! ###\nsub_1['article_id'] = sub_0['list'] + sub_1['article_id'] + ' ' + prediction_4['list']\nsub_1.article_id = sub_1.article_id.str.strip()\nsub_1.article_id = sub_1.article_id.str[:131]\nprint(sub_1.isna().values.any()) # true\nsub_1","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:40:44.648227Z","iopub.execute_input":"2022-05-09T06:40:44.648522Z","iopub.status.idle":"2022-05-09T06:40:48.63771Z","shell.execute_reply.started":"2022-05-09T06:40:44.648488Z","shell.execute_reply":"2022-05-09T06:40:48.636848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_0.index = sub_1.index\nsub_1.article_id.fillna('', inplace=True)\nsub_1['article_id'] = sub_1['article_id'] + sub_0['list']\nsub_1","metadata":{"execution":{"iopub.status.busy":"2022-05-09T07:23:43.456451Z","iopub.execute_input":"2022-05-09T07:23:43.457078Z","iopub.status.idle":"2022-05-09T07:23:43.778059Z","shell.execute_reply.started":"2022-05-09T07:23:43.457039Z","shell.execute_reply":"2022-05-09T07:23:43.777162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sub_1.isna().values.any()) # false\npd.set_option('display.max_colwidth', -1)\nprint(sub_1['article_id'])","metadata":{"execution":{"iopub.status.busy":"2022-05-09T07:24:06.235451Z","iopub.execute_input":"2022-05-09T07:24:06.235739Z","iopub.status.idle":"2022-05-09T07:24:06.548734Z","shell.execute_reply.started":"2022-05-09T07:24:06.235705Z","shell.execute_reply":"2022-05-09T07:24:06.547878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### SUBMISSION! ###\n#import os\n#os.remove(\"/kaggle/working/submission.csv\")\nsub_1.rename({'article_id':'prediction'}, inplace=True)\nsub_1.to_csv(f'submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T07:35:30.672066Z","iopub.execute_input":"2022-05-09T07:35:30.672877Z","iopub.status.idle":"2022-05-09T07:35:44.523348Z","shell.execute_reply.started":"2022-05-09T07:35:30.672839Z","shell.execute_reply":"2022-05-09T07:35:44.52242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Recommend 3 bestsellers that have discount as well","metadata":{}},{"cell_type":"code","source":"# First find the 12 bestsellers for each of the 5 index groups in the past year\nsimple_transactions_1y = transactions_1y[['article_id', 'customer_id', 'price']].copy()\nsimple_transactions_1y =\\\n    simple_transactions_1y.merge(articles_parquet[['article_id', 'index_group_name']], on='article_id', how='left')\nsimple_1y_bestsellers = simple_transactions_1y.groupby(['index_group_name', 'article_id']).price.count()\\\n    .reset_index(name='count').sort_values(['index_group_name', 'count'], ascending=[False, False]).copy()\nindex_group_bestsellers = simple_1y_bestsellers.groupby('index_group_name').head(12).copy()\n#index_group_bestsellers = index_group_bestsellers.merge(transactions_1y[['article_id', 'price']], on='article_id', how='left')\n#index_group_bestsellers[['index_group_name', 'article_id', 'price']].drop_duplicates(subset=['index_group_name', 'article_id']).reset_index(drop=True)\nindex_group_bestsellers = index_group_bestsellers[['index_group_name', 'article_id']].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:30:20.030512Z","iopub.execute_input":"2022-05-09T06:30:20.030782Z","iopub.status.idle":"2022-05-09T06:30:29.631207Z","shell.execute_reply.started":"2022-05-09T06:30:20.030753Z","shell.execute_reply":"2022-05-09T06:30:29.630439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Find the mode price for each article to represent its regular price in the past year.","metadata":{}},{"cell_type":"code","source":"simple_1y_price = simple_transactions_1y.groupby('article_id')['price'].value_counts()\\\n    .reset_index(name='price_count').sort_values(['article_id', 'price_count'], ascending=[False, False]).copy()\nsimple_1y_price = simple_1y_price.groupby('article_id').head(1)\nsimple_1y_price.rename({'price':'mode_price_y'}, axis=1, inplace=True)\nsimple_1y_price = simple_1y_price[['article_id', 'mode_price_y']].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:31:43.574733Z","iopub.execute_input":"2022-05-09T06:31:43.575049Z","iopub.status.idle":"2022-05-09T06:31:51.671504Z","shell.execute_reply.started":"2022-05-09T06:31:43.575014Z","shell.execute_reply":"2022-05-09T06:31:51.670627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Find the mode price of each article in the last week.","metadata":{}},{"cell_type":"code","source":"compare = index_group_bestsellers.merge(simple_1y_price, on='article_id', how='left').copy()\ncompare","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:31:59.507901Z","iopub.execute_input":"2022-05-09T06:31:59.508167Z","iopub.status.idle":"2022-05-09T06:31:59.537744Z","shell.execute_reply.started":"2022-05-09T06:31:59.508135Z","shell.execute_reply":"2022-05-09T06:31:59.536912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simple_1w_price = transactions_1w.groupby('article_id')['price'].value_counts()\\\n    .reset_index(name='price_count').sort_values(['article_id', 'price_count'], ascending=[False, False]).copy()\nsimple_1w_price = simple_1w_price.groupby('article_id').head(1)\nsimple_1w_price.rename({'price':'mode_price_w'}, axis=1, inplace=True)\nsimple_1w_price = simple_1w_price[['article_id', 'mode_price_w']].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:32:05.773342Z","iopub.execute_input":"2022-05-09T06:32:05.773861Z","iopub.status.idle":"2022-05-09T06:32:05.861753Z","shell.execute_reply.started":"2022-05-09T06:32:05.773824Z","shell.execute_reply":"2022-05-09T06:32:05.861057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compare it with its regular price computed above to see if it is on sale. It is interesting to see that when we multiply price_y by 0.9 and compare it with price_w, there is no 10% off item in the list, except article_id = 156231001 in Ladieswear. I decide to include this in our recommendations with the assumption that its big price fall would attract customers to buy it!","metadata":{}},{"cell_type":"code","source":"simple_price = compare.merge(simple_1w_price, on='article_id', how='left').copy()\nlist_3 = []\nfor article,price_w,price_y in zip(simple_price.article_id,simple_price.mode_price_w,simple_price.mode_price_y):\n    if price_w <= price_y:\n        list_3.append(str(article))\n    else:\n        list_3.append(float('nan'))\ncompare = pd.DataFrame({'index_group_name':compare.index_group_name, 'article_id':compare.article_id, 'result':list_3})\ncompare","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:32:13.079851Z","iopub.execute_input":"2022-05-09T06:32:13.08034Z","iopub.status.idle":"2022-05-09T06:32:13.1059Z","shell.execute_reply.started":"2022-05-09T06:32:13.080303Z","shell.execute_reply":"2022-05-09T06:32:13.105363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the table it is clear that for each index group, the top 3 bestsellers all have (previous week) price smaller than their corresponding regular price, so are good deals to customers. Another cause is proabably the overall price adjustment due to covid, but at least we have verified that our orginal recommendations don't have higher price, a factor which would hurt the volume of sales.\n\nThen for each list of 12 recommendations, find 3 articles with discount since last week. < 3 articles is acceptable. If there exists no articles with discount, just recommend the top 3 articles as usual.","metadata":{}},{"cell_type":"code","source":"predictions_3 = compare[['index_group_name', 'article_id']].groupby('index_group_name').head(3).copy()\npredictions_3 = predictions_3.reset_index(drop=True)\npredictions_3.iloc[8, 1] = '156231001'\npredictions_3 = predictions_3.groupby('index_group_name')['article_id'].apply(list).reset_index(name='list')\npre_predict_3 = transactions_parquet[['article_id', 'customer_id']].merge(articles_parquet[['article_id', 'index_group_name']], on='article_id', how='left').copy()\npre_predict_3 = pre_predict_3.groupby(['customer_id', 'index_group_name']).article_id.count()\\\n    .reset_index(name='count').sort_values(['customer_id', 'count'], ascending=[False, False]).copy()\npre_predict_3 = pre_predict_3.groupby('customer_id').head(1).copy()\npredictions_3 = pre_predict_3.merge(predictions_3, on='index_group_name', how='left')\npredictions_3 = predictions_3[['customer_id', 'list']].reset_index(drop=True)\npredictions_3","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:32:17.920759Z","iopub.execute_input":"2022-05-09T06:32:17.921037Z","iopub.status.idle":"2022-05-09T06:33:00.481911Z","shell.execute_reply.started":"2022-05-09T06:32:17.921006Z","shell.execute_reply":"2022-05-09T06:33:00.481321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_3['list'] = predictions_3['list'].apply(lambda x: ' 0'.join(map(str, x)))\npredictions_3.list = predictions_3.list.str.strip()\npredictions_3.list = predictions_3.list.str.zfill(32)\npredictions_3 = sub_parquet[['customer_id']].merge(predictions_3, on='customer_id', how='left')\npredictions_3","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:36:06.942482Z","iopub.execute_input":"2022-05-09T06:36:06.943194Z","iopub.status.idle":"2022-05-09T06:36:11.853071Z","shell.execute_reply.started":"2022-05-09T06:36:06.94315Z","shell.execute_reply":"2022-05-09T06:36:11.852245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sub_3.article_id.fillna('', inplace=True)\nsub_1['article_id'] = sub_1['article_id'] + ' ' + predictions_3['list']\nsub_1.article_id = sub_1.article_id.str.strip()\nsub_1","metadata":{"execution":{"iopub.status.busy":"2022-05-09T06:40:10.666533Z","iopub.execute_input":"2022-05-09T06:40:10.666856Z","iopub.status.idle":"2022-05-09T06:40:12.369055Z","shell.execute_reply.started":"2022-05-09T06:40:10.666817Z","shell.execute_reply":"2022-05-09T06:40:12.368092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Future Improvements:\n1. Use GPU and NVIDIA RAPIDS cudf to reduce runtime and memory usage so that\n   we can work on the entire transactions dataframe\n2. Tune the hyperparameters of our model to achieve optimal predictions\n3. Change to other combinations of the 4 strategies to achieve an optimal list\n   of recommendations\n4. We only removed duplicates between recommendations and purchase histories in the    end, but a better approach is to remove duplicates between 2 and 3, then 2,3 and 4.\n5. Content-Based Models: Use other attributes like the description of articles and build an NLP model\ntrain an item-based model find by aggregating a dataframe with necessary product attributes and rank with LightGBMRanker\n6. Time series analysis: check for seasonality and serial correlation\n7. Remove duplicates","metadata":{}}]}