{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Thank you for visit\n\n\n#### In this notebook, I want to show Example how to calculate MAP@12 \n\n#### If you like this notebook, upvote please 😉","metadata":{}},{"cell_type":"markdown","source":"### Version update (Feb 15, 2022)\n\n- I receive message from @hervind and @t88take that tell me my mistake about competition metrics.\n\n- So, I fixed this notebook.\n\n- Thank you.","metadata":{}},{"cell_type":"markdown","source":"### Version update (Feb 16, 2022)\n\n- I fixed this notebook again. 😂\n\n- I saw some discussions and learned from previous released notebooks.\n\n- If this will be useful for you, I'm happy too.","metadata":{}},{"cell_type":"markdown","source":"### Version update (Feb 26, 2022)\n\n- I fixed this notebook again. 😂\n\n- I dropped no purchase customer in valid term.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc\nimport os\nimport time\nimport random\nfrom tqdm.auto import tqdm","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_df(df):\n    print(df.shape)\n    display(df.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\nvisualize_df(transactions_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\ndel sub['prediction']; gc.collect()\nvisualize_df(sub)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transactions_train['t_dat'].unique()[-7:]\n\n# array(['2020-09-16', '2020-09-17', '2020-09-18', '2020-09-19',\n#       '2020-09-20', '2020-09-21', '2020-09-22'], dtype=object)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_start_date = '2020-09-16'\n\ntrain_data = transactions_train.query(f\"t_dat < '{val_start_date}'\").reset_index(drop=True)\nvalid_data = transactions_train.query(f\"t_dat >= '{val_start_date}'\").reset_index(drop=True)\n\nvisualize_df(train_data)\nvisualize_df(valid_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_unq = train_data.groupby('customer_id')['article_id'].apply(list).reset_index()\ntrain_unq['valid_pred'] = train_unq['article_id'].map(lambda x: '0'+' 0'.join(str(x)[1:-1].split(', ')))\nvisualize_df(train_unq)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_unq = valid_data.groupby('customer_id')['article_id'].apply(list).reset_index()\nvalid_unq['valid_true'] = valid_unq['article_id'].map(lambda x: '0'+' 0'.join(str(x)[1:-1].split(', ')))\nvisualize_df(valid_unq)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged = pd.merge(sub, train_unq, on='customer_id', how='left').fillna('')\nmerged = pd.merge(merged, valid_unq, on='customer_id', how='left').fillna('')\n\ndel merged['article_id_x'], merged['article_id_y']; gc.collect()\nmerged.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove customers who made no purchases \n\nmerged = merged[merged['valid_true']!=''].reset_index(drop=True)\nprint(merged.shape)\nmerged.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/c/h-and-m-personalized-fashion-recommendations/discussion/306007\n# https://github.com/benhamner/Metrics/blob/master/Python/ml_metrics/average_precision.py\n\n\ndef apk(actual, predicted, k=10):\n    \"\"\"\n    Computes the average precision at k.\n    This function computes the average prescision at k between two lists of\n    items.\n    Parameters\n    ----------\n    actual : list\n             A list of elements that are to be predicted (order doesn't matter)\n    predicted : list\n                A list of predicted elements (order does matter)\n    k : int, optional\n        The maximum number of predicted elements\n    Returns\n    -------\n    score : double\n            The average precision at k over the input lists\n    \"\"\"\n    if len(predicted)>k:\n        predicted = predicted[:k]\n\n    score = 0.0\n    num_hits = 0.0\n\n    for i,p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            num_hits += 1.0\n            score += num_hits / (i+1.0)\n\n    # remove this case in advance\n    # if not actual:\n    #     return 0.0\n\n    return score / min(len(actual), k)\n\n\ndef mapk(actual, predicted, k=10):\n    \"\"\"\n    Computes the mean average precision at k.\n    This function computes the mean average prescision at k between two lists\n    of lists of items.\n    Parameters\n    ----------\n    actual : list\n             A list of lists of elements that are to be predicted \n             (order doesn't matter in the lists)\n    predicted : list\n                A list of lists of predicted elements\n                (order matters in the lists)\n    k : int, optional\n        The maximum number of predicted elements\n    Returns\n    -------\n    score : double\n            The mean average precision at k over the input lists\n    \"\"\"\n    return np.mean([apk(a,p,k) for a,p in zip(actual, predicted)])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()\n\nmapk(\n    merged['valid_true'].map(lambda x: x.split()), \n    merged['valid_pred'].map(lambda x: x.split()), \n    k=12\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### CV is 0.0029\n\n#### Actual LB score is 0.001XX","metadata":{}},{"cell_type":"code","source":"sub = merged[['customer_id', 'valid_pred']].copy()\nsub.columns = ['customer_id', 'prediction']\nprint(sub.shape)\n\nsub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}