{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Thanks for Visiting! ","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I would like to share how my approach to evaluate generated candidates ","metadata":{}},{"cell_type":"markdown","source":"# 5 Things to be evaluated","metadata":{}},{"cell_type":"markdown","source":"\n\n1. Maximum MAP12: what if the candidate is already sorted by the relevancy (Ranking model is optimum)\n2. Full covered customer: # customer that all their purchased articles inside the candidates\n3. Not covered customer: # customer that nothing in their purchased articles inside the candidates \n4. Candidate multiplier: Ratio between rows in candidate dataframe and validation dataframe \n5. total unique article id: The lesser the better to optimize memory ","metadata":{}},{"cell_type":"markdown","source":"# Lets Code ","metadata":{}},{"cell_type":"markdown","source":"Lets start from the started one. \n\nHere we just generate top 100 items based on last week before validation week to be candidates items of all customer ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport gc\n\nimport cudf","metadata":{"execution":{"iopub.status.busy":"2022-04-17T02:10:21.739155Z","iopub.execute_input":"2022-04-17T02:10:21.739455Z","iopub.status.idle":"2022-04-17T02:10:25.166162Z","shell.execute_reply.started":"2022-04-17T02:10:21.739379Z","shell.execute_reply":"2022-04-17T02:10:25.165418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n                            usecols= ['t_dat', 'customer_id', 'article_id'], \n                            dtype={'article_id': 'int32', 't_dat': 'string', 'customer_id': 'string'})\ntransactions['customer_id'] = transactions['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n\nsubmission = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv',\n                        usecols=['customer_id'])\nsubmission['customer_id'] = submission['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n\n\ntrain_start_date = '2020-09-09'\nvalid_start_date = '2020-09-16'\n\ntrain_df = transactions.loc[((transactions['t_dat'] >= train_start_date) & (transactions['t_dat'] < valid_start_date)), \n                           ['customer_id', 'article_id']].drop_duplicates().reset_index(drop=True)\nvalid_df = transactions.loc[transactions['t_dat'] >= valid_start_date,\n                           ['customer_id', 'article_id']].drop_duplicates().reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T02:17:25.398168Z","iopub.execute_input":"2022-04-17T02:17:25.398417Z","iopub.status.idle":"2022-04-17T02:17:28.103790Z","shell.execute_reply.started":"2022-04-17T02:17:25.398390Z","shell.execute_reply":"2022-04-17T02:17:28.103048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_cust = valid_df['customer_id'].unique().to_pandas().to_list()\ntop_100_popular = train_df['article_id'].value_counts()[:100].index.to_pandas().to_list()\n\ncandidate_df = cudf.DataFrame({'customer_id': np.repeat(valid_cust, len(top_100_popular)),\n                              'article_id': top_100_popular * len(valid_cust),\n                              })\n\nprint(len(valid_cust), len(top_100_popular), len(candidate_df))","metadata":{"execution":{"iopub.status.busy":"2022-04-17T02:31:01.916853Z","iopub.execute_input":"2022-04-17T02:31:01.917548Z","iopub.status.idle":"2022-04-17T02:31:02.503195Z","shell.execute_reply.started":"2022-04-17T02:31:01.917516Z","shell.execute_reply":"2022-04-17T02:31:02.502416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def candidate_score(candidate_df, valid_df):\n    both_df = candidate_df.merge(valid_df, on = ['customer_id', 'article_id'], how = 'inner')\n\n    # if the customer have >=12 just count 12 \n    trgt_cnt_df = valid_df.groupby('customer_id', as_index = False).agg({'article_id' :'count'}).\\\n        rename(columns = {'article_id':'trgt_cnt'})\n    trgt_cnt_df.loc[trgt_cnt_df['trgt_cnt']>= 12, 'trgt_cnt'] = 12 \n    both_cnt_df = both_df.groupby('customer_id', as_index = False).agg({'article_id' :'count'}).\\\n        rename(columns = {'article_id':'both_cnt'})\n    both_cnt_df.loc[both_cnt_df['both_cnt']>= 12, 'both_cnt'] = 12 \n\n    trgt_cnt_df = trgt_cnt_df.merge(both_cnt_df, on = 'customer_id', how = 'left') \n    trgt_cnt_df.fillna(0, inplace = True)\n    # assume it is optimally sorted \n    trgt_cnt_df['AP12'] = trgt_cnt_df['both_cnt'] / trgt_cnt_df['trgt_cnt']\n    max_map12 = trgt_cnt_df['AP12'].mean()\n\n    full_covered_cust = len(trgt_cnt_df.loc[trgt_cnt_df['AP12'] == 1])\n    not_covered_cust = len(trgt_cnt_df.loc[trgt_cnt_df['AP12'] == 0])\n    num_target_cust = len(trgt_cnt_df)\n\n    num_candidate = len(candidate_df)\n    num_target = len(valid_df)\n    num_unq_artc = len(candidate_df['article_id'].unique())\n    \n\n    print(f\"MAX MAP12: {round(max_map12, 4)}\")\n    print(f\"Full Covered Customer: {round(full_covered_cust / num_target_cust, 4)} ({full_covered_cust} / {num_target_cust}) \")\n    print(f\"Not Covered Customer: {round(not_covered_cust / num_target_cust, 4)} ({not_covered_cust} / {num_target_cust}) \")\n    print(f\"Candidate multiplier: {round(num_candidate / num_target, 4)} ({num_candidate} / {num_target})\")\n    print(f\"Unique Article Id: {num_unq_artc}\")\n    \n    return max_map12, full_covered_cust / num_target_cust, not_covered_cust / num_target_cust, num_candidate / num_target ,num_unq_artc","metadata":{"execution":{"iopub.status.busy":"2022-04-17T02:36:16.244366Z","iopub.execute_input":"2022-04-17T02:36:16.244639Z","iopub.status.idle":"2022-04-17T02:36:16.255203Z","shell.execute_reply.started":"2022-04-17T02:36:16.244611Z","shell.execute_reply":"2022-04-17T02:36:16.254331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"candidate_score(candidate_df, valid_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-17T02:36:16.471054Z","iopub.execute_input":"2022-04-17T02:36:16.471301Z","iopub.status.idle":"2022-04-17T02:36:16.520225Z","shell.execute_reply.started":"2022-04-17T02:36:16.471275Z","shell.execute_reply":"2022-04-17T02:36:16.519401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}