{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview","metadata":{}},{"cell_type":"markdown","source":"In competition, it takes a lot of time to pre-process or training because too many items of information are included.\n\nSo I try to find an efficient set of items for making predictions.\n\n*Referece Notebook*\n- Byfone: https://www.kaggle.com/byfone/h-m-trending-products-weekly","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-10T17:02:04.126791Z","iopub.execute_input":"2022-03-10T17:02:04.127328Z","iopub.status.idle":"2022-03-10T17:02:05.161772Z","shell.execute_reply.started":"2022-03-10T17:02:04.127236Z","shell.execute_reply":"2022-03-10T17:02:05.160989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Dataset","metadata":{}},{"cell_type":"code","source":"articles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\ntransactions = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', usecols=['t_dat', 'customer_id', 'article_id'])","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:02:19.314037Z","iopub.execute_input":"2022-03-10T17:02:19.314334Z","iopub.status.idle":"2022-03-10T17:03:25.292416Z","shell.execute_reply.started":"2022-03-10T17:02:19.314305Z","shell.execute_reply":"2022-03-10T17:03:25.291404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article2idx = dict(zip(articles[\"article_id\"], articles.index))\nidx2article = dict(zip(articles.index, articles[\"article_id\"]))\ndel articles","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:03:25.295220Z","iopub.execute_input":"2022-03-10T17:03:25.295871Z","iopub.status.idle":"2022-03-10T17:03:25.393092Z","shell.execute_reply.started":"2022-03-10T17:03:25.295817Z","shell.execute_reply":"2022-03-10T17:03:25.391967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions[\"article_id\"] = transactions[\"article_id\"].map(lambda x: article2idx[x])","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:03:34.198616Z","iopub.execute_input":"2022-03-10T17:03:34.199259Z","iopub.status.idle":"2022-03-10T17:04:27.005542Z","shell.execute_reply.started":"2022-03-10T17:03:34.199214Z","shell.execute_reply":"2022-03-10T17:04:27.004713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:04:54.316196Z","iopub.execute_input":"2022-03-10T17:04:54.316498Z","iopub.status.idle":"2022-03-10T17:04:54.450271Z","shell.execute_reply.started":"2022-03-10T17:04:54.316467Z","shell.execute_reply":"2022-03-10T17:04:54.449694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split Train/Test","metadata":{}},{"cell_type":"markdown","source":"before predict, split the train/test dataset","metadata":{}},{"cell_type":"code","source":"transactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\nprint(transactions['t_dat'].max())","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:05:02.279997Z","iopub.execute_input":"2022-03-10T17:05:02.280492Z","iopub.status.idle":"2022-03-10T17:05:08.409666Z","shell.execute_reply.started":"2022-03-10T17:05:02.280441Z","shell.execute_reply":"2022-03-10T17:05:08.408823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = transactions.query(\"t_dat<='2020-09-15'\").reset_index(drop=True)\ntest = transactions.query(\"t_dat>'2020-09-15'\").reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:05:14.537623Z","iopub.execute_input":"2022-03-10T17:05:14.537960Z","iopub.status.idle":"2022-03-10T17:05:16.758441Z","shell.execute_reply.started":"2022-03-10T17:05:14.537928Z","shell.execute_reply":"2022-03-10T17:05:16.757533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)\nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:05:16.760295Z","iopub.execute_input":"2022-03-10T17:05:16.760554Z","iopub.status.idle":"2022-03-10T17:05:16.765000Z","shell.execute_reply.started":"2022-03-10T17:05:16.760522Z","shell.execute_reply":"2022-03-10T17:05:16.763959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Method","metadata":{}},{"cell_type":"markdown","source":"## 1.Prediction method using geometric distribution","metadata":{}},{"cell_type":"code","source":"summary = train.groupby('article_id')['t_dat'].agg(['min', 'max', 'count', 'nunique']).reset_index()\nsummary = summary.rename(columns={'min': 'min_dat', 'max': 'max_dat', 'count':'total_sales', 'nunique':'unique_dat'})","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:05:21.950118Z","iopub.execute_input":"2022-03-10T17:05:21.950563Z","iopub.status.idle":"2022-03-10T17:05:30.847723Z","shell.execute_reply.started":"2022-03-10T17:05:21.950530Z","shell.execute_reply":"2022-03-10T17:05:30.846954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary['diff_dat'] = (summary['max_dat'] - summary['min_dat']).dt.days \nsummary['diff_dat'] = pd.TimedeltaIndex(summary['diff_dat'] + 1, unit='D').days\n\nlast_tdat = train['t_dat'].max()\nsummary['last_diff_dat'] = (last_tdat - summary['max_dat']).dt.days","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:05:30.849088Z","iopub.execute_input":"2022-03-10T17:05:30.849456Z","iopub.status.idle":"2022-03-10T17:05:30.973285Z","shell.execute_reply.started":"2022-03-10T17:05:30.849427Z","shell.execute_reply":"2022-03-10T17:05:30.972444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary['avg_sales'] = summary['total_sales'] / summary['unique_dat']\nsummary['tdat_ratio'] = summary['unique_dat'] / summary['diff_dat']\n\nsummary['daily_sales'] = summary['avg_sales'] * summary['tdat_ratio']","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:05:30.974322Z","iopub.execute_input":"2022-03-10T17:05:30.975121Z","iopub.status.idle":"2022-03-10T17:05:30.984711Z","shell.execute_reply.started":"2022-03-10T17:05:30.975075Z","shell.execute_reply":"2022-03-10T17:05:30.983760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(summary['daily_sales'], kind='kde')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T16:34:15.523894Z","iopub.execute_input":"2022-03-10T16:34:15.524178Z","iopub.status.idle":"2022-03-10T16:34:16.357943Z","shell.execute_reply.started":"2022-03-10T16:34:15.524148Z","shell.execute_reply":"2022-03-10T16:34:16.357052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T16:34:16.359114Z","iopub.execute_input":"2022-03-10T16:34:16.359337Z","iopub.status.idle":"2022-03-10T16:34:16.379096Z","shell.execute_reply.started":"2022-03-10T16:34:16.359308Z","shell.execute_reply":"2022-03-10T16:34:16.378512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We calculated the average sales and the ratio at which transactions occurred.\n\nMultiply the above two variables to get the daily sales.\n\n**`daily_sales` = `avg_sales` * `tdat_ratio`**\n\nDaily Sales is the average number of articles that can be sold per day.\n\nHowever, adjustments are required for articles that have not traded until recently","metadata":{}},{"cell_type":"code","source":"from scipy.stats import geom\n\ndef geom_func(p, n):\n    n //= 7\n    if n==0:\n        return p\n    return geom(p).pmf(n)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:05:30.986798Z","iopub.execute_input":"2022-03-10T17:05:30.987235Z","iopub.status.idle":"2022-03-10T17:05:31.000039Z","shell.execute_reply.started":"2022-03-10T17:05:30.987203Z","shell.execute_reply":"2022-03-10T17:05:30.999131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary['alpha'] = summary.apply(lambda x: geom_func(x['tdat_ratio'], x['last_diff_dat']), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:05:31.001772Z","iopub.execute_input":"2022-03-10T17:05:31.002111Z","iopub.status.idle":"2022-03-10T17:06:49.595088Z","shell.execute_reply.started":"2022-03-10T17:05:31.002068Z","shell.execute_reply":"2022-03-10T17:06:49.593871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary['pred_sales'] = summary['alpha'] * summary['daily_sales']","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:06:49.596960Z","iopub.execute_input":"2022-03-10T17:06:49.597287Z","iopub.status.idle":"2022-03-10T17:06:49.604286Z","shell.execute_reply.started":"2022-03-10T17:06:49.597241Z","shell.execute_reply":"2022-03-10T17:06:49.603214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*To avoid confusion, the expression sales does not refer to actual sales.*","metadata":{}},{"cell_type":"code","source":"sns.displot(summary['pred_sales'], kind='kde')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T16:35:34.839952Z","iopub.execute_input":"2022-03-10T16:35:34.840161Z","iopub.status.idle":"2022-03-10T16:35:35.623329Z","shell.execute_reply.started":"2022-03-10T16:35:34.840135Z","shell.execute_reply":"2022-03-10T16:35:35.622488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary = summary.sort_values(by='pred_sales', ascending=False).reset_index(drop=True)\nsummary.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T16:35:35.625559Z","iopub.execute_input":"2022-03-10T16:35:35.625810Z","iopub.status.idle":"2022-03-10T16:35:35.676342Z","shell.execute_reply.started":"2022-03-10T16:35:35.625782Z","shell.execute_reply":"2022-03-10T16:35:35.675605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sales = summary.query('pred_sales>=1')['article_id'].values\ndaily_sales = summary.query('daily_sales>=1')['article_id'].values","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:06:49.607665Z","iopub.execute_input":"2022-03-10T17:06:49.607986Z","iopub.status.idle":"2022-03-10T17:06:49.634191Z","shell.execute_reply.started":"2022-03-10T17:06:49.607940Z","shell.execute_reply":"2022-03-10T17:06:49.633022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Weekly Sales","metadata":{}},{"cell_type":"markdown","source":"In Weekly Sales, we get `quotient`\n\n- `quotient` : last_week_sales / ldbw_sales\n\nHow to create a variable is detailed in [Byfone's Notebook](https://www.kaggle.com/byfone/h-m-trending-products-weekly)","metadata":{}},{"cell_type":"code","source":"last_ts = train['t_dat'].max()\n\n# df['ldbw'] = df['t_dat'].progress_apply(lambda d: last_ts - (last_ts - d).floor('7D'))\ntrain['offset_dat'] = (last_ts - train['t_dat']).dt.floor('7D')\ntrain['ldbw'] = last_ts - train['offset_dat']\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:06:49.635940Z","iopub.execute_input":"2022-03-10T17:06:49.636261Z","iopub.status.idle":"2022-03-10T17:06:51.442881Z","shell.execute_reply.started":"2022-03-10T17:06:49.636216Z","shell.execute_reply":"2022-03-10T17:06:51.441963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weekly_sales = train.groupby(['ldbw', 'article_id']).t_dat.count().reset_index(name='count')\nweekly_sales.tail()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:06:51.444085Z","iopub.execute_input":"2022-03-10T17:06:51.444402Z","iopub.status.idle":"2022-03-10T17:06:55.208356Z","shell.execute_reply.started":"2022-03-10T17:06:51.444371Z","shell.execute_reply":"2022-03-10T17:06:55.207317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.merge(weekly_sales, on=['ldbw', 'article_id'], how='left')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:06:55.210626Z","iopub.execute_input":"2022-03-10T17:06:55.211546Z","iopub.status.idle":"2022-03-10T17:07:02.634330Z","shell.execute_reply.started":"2022-03-10T17:06:55.211492Z","shell.execute_reply":"2022-03-10T17:07:02.633394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weekly_sales = weekly_sales.set_index('article_id')\n\ntrain = train.merge(weekly_sales.loc[weekly_sales['ldbw']==last_ts, ['count']],\n                    how='left',\n                    on='article_id', \n                    suffixes=(\"\", \"_targ\"))\n\n# last week sales\ntrain['count_targ'].fillna(0, inplace=True)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:07:02.635885Z","iopub.execute_input":"2022-03-10T17:07:02.636206Z","iopub.status.idle":"2022-03-10T17:07:09.455729Z","shell.execute_reply.started":"2022-03-10T17:07:02.636162Z","shell.execute_reply":"2022-03-10T17:07:09.454718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# last_week_sales / ldbw_sales\ntrain['quotient'] = train['count_targ'] / train['count']\ntrain = train.sort_values(by='quotient', ascending=False).reset_index(drop=True)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:07:09.457313Z","iopub.execute_input":"2022-03-10T17:07:09.457630Z","iopub.status.idle":"2022-03-10T17:07:22.529889Z","shell.execute_reply.started":"2022-03-10T17:07:09.457586Z","shell.execute_reply":"2022-03-10T17:07:22.529026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lw_sales = set(train.query('quotient>=1')['article_id'].values)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:11:32.547272Z","iopub.execute_input":"2022-03-10T17:11:32.547891Z","iopub.status.idle":"2022-03-10T17:11:33.272506Z","shell.execute_reply.started":"2022-03-10T17:11:32.547838Z","shell.execute_reply":"2022-03-10T17:11:33.271545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Total Article Size: {train.article_id.nunique()}')\nprint('-' * 80)\nprint(f'Pred Sales Candidate Size: {len(pred_sales)}')\nprint(f'Daily Sales Candidate Size: {len(daily_sales)}')\nprint(f'Last Weekly Sales(Qutotient) Candidate Size: {len(lw_sales)}')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:07:23.322412Z","iopub.execute_input":"2022-03-10T17:07:23.322669Z","iopub.status.idle":"2022-03-10T17:07:23.548927Z","shell.execute_reply.started":"2022-03-10T17:07:23.322626Z","shell.execute_reply":"2022-03-10T17:07:23.547968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"from collections import Counter\n\ntest_sales = Counter(test['article_id'].values)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:07:35.615886Z","iopub.execute_input":"2022-03-10T17:07:35.616171Z","iopub.status.idle":"2022-03-10T17:07:35.694201Z","shell.execute_reply.started":"2022-03-10T17:07:35.616143Z","shell.execute_reply":"2022-03-10T17:07:35.693171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_size = len(test_sales.keys())\nprint(f'Test Articles Count: {test_size}')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:07:39.350294Z","iopub.execute_input":"2022-03-10T17:07:39.350584Z","iopub.status.idle":"2022-03-10T17:07:39.356290Z","shell.execute_reply.started":"2022-03-10T17:07:39.350552Z","shell.execute_reply":"2022-03-10T17:07:39.355412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Matched Articles Count: {len(set(test_sales.keys()).intersection(set(pred_sales)))}')\nprint(f'Matched Articles Ratio: {len(set(test_sales.keys()).intersection(set(pred_sales))) / test_size}')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:07:41.165634Z","iopub.execute_input":"2022-03-10T17:07:41.165984Z","iopub.status.idle":"2022-03-10T17:07:41.183524Z","shell.execute_reply.started":"2022-03-10T17:07:41.165947Z","shell.execute_reply":"2022-03-10T17:07:41.182592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Matched Articles Count: {len(set(test_sales.keys()).intersection(set(daily_sales)))}')\nprint(f'Matched Articles Ratio: {len(set(test_sales.keys()).intersection(set(daily_sales))) / test_size}')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:07:42.465498Z","iopub.execute_input":"2022-03-10T17:07:42.467578Z","iopub.status.idle":"2022-03-10T17:07:42.596110Z","shell.execute_reply.started":"2022-03-10T17:07:42.467531Z","shell.execute_reply":"2022-03-10T17:07:42.587134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Matched Articles Count: {len(set(test_sales.keys()).intersection(set(lw_sales)))}')\nprint(f'Matched Articles Ratio: {len(set(test_sales.keys()).intersection(set(lw_sales))) / test_size}')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:08:48.371438Z","iopub.execute_input":"2022-03-10T17:08:48.371769Z","iopub.status.idle":"2022-03-10T17:08:48.386595Z","shell.execute_reply.started":"2022-03-10T17:08:48.371736Z","shell.execute_reply":"2022-03-10T17:08:48.385964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`lw_sales` can be seen that the matched articles ratio is quite high(0.77%).\n\nConsidering the size of each candidate, `pred_sales` can do a lot of matching even with a relatively small size.\n\nPerhaps if you increase the size a little more, you can increase the matching ratio.","metadata":{}},{"cell_type":"code","source":"pred_sales = summary.query('pred_sales>=0.1')['article_id'].values\nprint(f'Pred Sales Candidate Size: {len(pred_sales)}')\n\nprint(f'Matched Articles Count: {len(set(test_sales.keys()).intersection(set(pred_sales)))}')\nprint(f'Matched Articles Ratio: {len(set(test_sales.keys()).intersection(set(pred_sales))) / test_size}')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:07:47.114772Z","iopub.execute_input":"2022-03-10T17:07:47.115111Z","iopub.status.idle":"2022-03-10T17:07:47.148123Z","shell.execute_reply.started":"2022-03-10T17:07:47.115075Z","shell.execute_reply":"2022-03-10T17:07:47.147331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**when pred_sales set to 0.1, 80% of the articles could be matched to the test data even with about 20,000 candidate sizes.**\n\nThis can increase the accuracy of the model by effectively reducing the number of articles to be considered.","metadata":{}},{"cell_type":"code","source":"pred_sales = summary.query('pred_sales>=0.01')['article_id'].values\nprint(f'Pred Sales Candidate Size: {len(pred_sales)}')\n\nprint(f'Matched Articles Count: {len(set(test_sales.keys()).intersection(set(pred_sales)))}')\nprint(f'Matched Articles Ratio: {len(set(test_sales.keys()).intersection(set(pred_sales))) / test_size}')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:15:19.264894Z","iopub.execute_input":"2022-03-10T17:15:19.265428Z","iopub.status.idle":"2022-03-10T17:15:19.300523Z","shell.execute_reply.started":"2022-03-10T17:15:19.265394Z","shell.execute_reply":"2022-03-10T17:15:19.299842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**when pred_sales set to 0.01, 92% of the articles could be matched to the test data even with about 30,000 candidate sizes.**","metadata":{}},{"cell_type":"code","source":"lw_sales = set(train.query('quotient>=0.1')['article_id'].values)\nprint(f'Last Weekly Sales(Qutotient) Candidate Size: {len(lw_sales)}')\n\nprint(f'Matched Articles Count: {len(set(test_sales.keys()).intersection(set(lw_sales)))}')\nprint(f'Matched Articles Ratio: {len(set(test_sales.keys()).intersection(set(lw_sales))) / test_size}')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T17:19:45.290408Z","iopub.execute_input":"2022-03-10T17:19:45.290781Z","iopub.status.idle":"2022-03-10T17:19:48.324995Z","shell.execute_reply.started":"2022-03-10T17:19:45.290741Z","shell.execute_reply":"2022-03-10T17:19:48.323930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Additionally, there was no difference when the ratio were adjusted for the `lw_sales`.","metadata":{}},{"cell_type":"markdown","source":"## Top100 Articles","metadata":{}},{"cell_type":"markdown","source":"Basically, the evaluation method considering the order is [ndcg](https://en.wikipedia.org/wiki/Discounted_cumulative_gain).\n\nHowever, since the evaluation method in the competition is a map, we will check whether it is actually in the top 100.","metadata":{}},{"cell_type":"code","source":"top100 = sorted(test_sales.items(), key=lambda x: -x[1])[:100]\n\ndf = pd.DataFrame(top100, columns=['article_id', 'sales'])","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:03:37.408131Z","iopub.execute_input":"2022-03-09T18:03:37.40894Z","iopub.status.idle":"2022-03-09T18:03:37.421279Z","shell.execute_reply.started":"2022-03-09T18:03:37.408889Z","shell.execute_reply":"2022-03-09T18:03:37.420564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary['rank'] = summary.index\n\ndf = df.merge(summary[['article_id', 'rank']], how='left', on='article_id')","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:08:54.99686Z","iopub.execute_input":"2022-03-09T18:08:54.997192Z","iopub.status.idle":"2022-03-09T18:08:55.028996Z","shell.execute_reply.started":"2022-03-09T18:08:54.997154Z","shell.execute_reply":"2022-03-09T18:08:55.02836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:09:19.07766Z","iopub.execute_input":"2022-03-09T18:09:19.078147Z","iopub.status.idle":"2022-03-09T18:09:19.089725Z","shell.execute_reply.started":"2022-03-09T18:09:19.0781Z","shell.execute_reply":"2022-03-09T18:09:19.088973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"index 18, 19 is Missing Rank(NaN)\n\nMaybe, The missing values will be the first articles that appears at the test periods","metadata":{}},{"cell_type":"code","source":"df.loc[df['rank'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:13:33.438384Z","iopub.execute_input":"2022-03-09T18:13:33.43869Z","iopub.status.idle":"2022-03-09T18:13:33.447764Z","shell.execute_reply.started":"2022-03-09T18:13:33.438656Z","shell.execute_reply":"2022-03-09T18:13:33.447186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.query('article_id==105222')['t_dat'].min()","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:13:08.691716Z","iopub.execute_input":"2022-03-09T18:13:08.69197Z","iopub.status.idle":"2022-03-09T18:13:08.914313Z","shell.execute_reply.started":"2022-03-09T18:13:08.691943Z","shell.execute_reply":"2022-03-09T18:13:08.913496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will check only 95 items, excluding missing values.","metadata":{}},{"cell_type":"code","source":"df = df.loc[df['rank'].notna()].reset_index(drop=True)\n\ndf['correct'] = df['rank'].map(lambda x: 1 if x<=100 else 0)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:18:52.077227Z","iopub.execute_input":"2022-03-09T18:18:52.077488Z","iopub.status.idle":"2022-03-09T18:18:52.084209Z","shell.execute_reply.started":"2022-03-09T18:18:52.077453Z","shell.execute_reply":"2022-03-09T18:18:52.083423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['correct'].sum() / len(df)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:19:39.083197Z","iopub.execute_input":"2022-03-09T18:19:39.083607Z","iopub.status.idle":"2022-03-09T18:19:39.090484Z","shell.execute_reply.started":"2022-03-09T18:19:39.083568Z","shell.execute_reply":"2022-03-09T18:19:39.089867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The top 95 items were matched at a rate of 45% out of 95.\nThis could be used as general_pred.","metadata":{}},{"cell_type":"markdown","source":"# **If you find this note book helpful, please upvote!!**","metadata":{}}]}