{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook provides ranking baseline that uses item, user features and lightgbm as the ranker model. Code for preparing item features [this](https://www.kaggle.com/alexvishnevskiy/ranking-item-features), code for preparing user features [this](https://www.kaggle.com/alexvishnevskiy/ranking-user-features). Some code is taken from [this repo](https://github.com/radekosmulski/personalized_fashion_recs).","metadata":{"id":"70bkjrfStDfM"}},{"cell_type":"code","source":"from lightgbm.sklearn import LGBMRanker\nfrom datetime import timedelta\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nfrom tqdm import tqdm","metadata":{"id":"bU6vwaTcDBVB","execution":{"iopub.status.busy":"2022-02-27T20:42:26.109634Z","iopub.execute_input":"2022-02-27T20:42:26.110156Z","iopub.status.idle":"2022-02-27T20:42:28.07762Z","shell.execute_reply.started":"2022-02-27T20:42:26.110122Z","shell.execute_reply":"2022-02-27T20:42:28.076703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load all data","metadata":{}},{"cell_type":"code","source":"user_features = pd.read_parquet('../input/ranking-features/user_features.parquet')\nitem_features = pd.read_parquet('../input/ranking-features/item_features.parquet')\ntransactions_df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ntransactions_df.t_dat = pd.to_datetime( transactions_df.t_dat )","metadata":{"id":"wcdV4WcbDP_e","execution":{"iopub.status.busy":"2022-02-27T20:48:59.283035Z","iopub.execute_input":"2022-02-27T20:48:59.283581Z","iopub.status.idle":"2022-02-27T20:50:16.76493Z","shell.execute_reply.started":"2022-02-27T20:48:59.283525Z","shell.execute_reply":"2022-02-27T20:50:16.763925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Last 4 weeks of transactions will be used as a baseline.","metadata":{"id":"hckIzN4qswV2"}},{"cell_type":"code","source":"df_4w = transactions_df[transactions_df['t_dat'] >= pd.to_datetime('2020-08-24')].copy()\ndf_3w = transactions_df[transactions_df['t_dat'] >= pd.to_datetime('2020-08-31')].copy()\ndf_2w = transactions_df[transactions_df['t_dat'] >= pd.to_datetime('2020-09-07')].copy()\ndf_1w = transactions_df[transactions_df['t_dat'] >= pd.to_datetime('2020-09-15')].copy()","metadata":{"id":"IYyTDVwLdaTr","execution":{"iopub.status.busy":"2022-02-27T20:43:46.581936Z","iopub.execute_input":"2022-02-27T20:43:46.582273Z","iopub.status.idle":"2022-02-27T20:43:47.256329Z","shell.execute_reply.started":"2022-02-27T20:43:46.58223Z","shell.execute_reply":"2022-02-27T20:43:47.255571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Factorize all categorical features","metadata":{"id":"p1OGJZ4us24-"}},{"cell_type":"code","source":"user_features[['club_member_status', 'fashion_news_frequency']] = (\n                   user_features[['club_member_status', 'fashion_news_frequency']]\n                   .apply(lambda x: pd.factorize(x)[0])\n).astype('int8')","metadata":{"id":"ZHnAsBk1HNNu","execution":{"iopub.status.busy":"2022-02-27T20:43:47.257622Z","iopub.execute_input":"2022-02-27T20:43:47.258652Z","iopub.status.idle":"2022-02-27T20:43:47.564324Z","shell.execute_reply.started":"2022-02-27T20:43:47.258601Z","shell.execute_reply":"2022-02-27T20:43:47.563505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Merge user, item features to transactions.","metadata":{"id":"hLGSgcccs9zr"}},{"cell_type":"code","source":"transactions_df = (\n    transactions_df\n    .merge(user_features, on = ('customer_id'))\n    .merge(item_features, on = ('article_id'))\n)\ntransactions_df.sort_values(['t_dat', 'customer_id'], inplace=True)","metadata":{"id":"Jf0_LM-JHrUR","execution":{"iopub.status.busy":"2022-02-27T20:50:16.766736Z","iopub.execute_input":"2022-02-27T20:50:16.767018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for simplicity let's take only 1M rows\nN_ROWS = 1_000_000\n\ntrain = transactions_df.loc[ transactions_df.t_dat <= pd.to_datetime('2020-09-15') ].iloc[:N_ROWS]\nvalid = transactions_df.loc[ transactions_df.t_dat >= pd.to_datetime('2020-09-16') ]","metadata":{"id":"anyNkYukDQB1","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#delete transactions to save memory\ndel transactions_df","metadata":{"id":"B--Vk-g3dvfc","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, valid.shape","metadata":{"id":"so86O-mCE4M_","outputId":"c9fdad26-c63d-452e-cd7f-ad17d7fd44b6","execution":{"iopub.status.busy":"2022-02-27T20:47:11.772647Z","iopub.execute_input":"2022-02-27T20:47:11.772911Z","iopub.status.idle":"2022-02-27T20:47:11.790616Z","shell.execute_reply.started":"2022-02-27T20:47:11.772877Z","shell.execute_reply":"2022-02-27T20:47:11.789754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare candidates","metadata":{"id":"iQpAr_iNfkNi"}},{"cell_type":"code","source":"purchase_dict_4w = {}\n\nfor i,x in enumerate(zip(df_4w['customer_id'], df_4w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_4w:\n        purchase_dict_4w[cust_id] = {}\n    \n    if art_id not in purchase_dict_4w[cust_id]:\n        purchase_dict_4w[cust_id][art_id] = 0\n    \n    purchase_dict_4w[cust_id][art_id] += 1\n\ndummy_list_4w = list((df_4w['article_id'].value_counts()).index)[:12]","metadata":{"id":"b12XmH1SfoPR","execution":{"iopub.status.busy":"2022-02-27T20:47:11.792248Z","iopub.execute_input":"2022-02-27T20:47:11.792686Z","iopub.status.idle":"2022-02-27T20:47:13.540929Z","shell.execute_reply.started":"2022-02-27T20:47:11.792648Z","shell.execute_reply":"2022-02-27T20:47:13.539831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_3w = {}\n\nfor i,x in enumerate(zip(df_3w['customer_id'], df_3w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_3w:\n        purchase_dict_3w[cust_id] = {}\n    \n    if art_id not in purchase_dict_3w[cust_id]:\n        purchase_dict_3w[cust_id][art_id] = 0\n    \n    purchase_dict_3w[cust_id][art_id] += 1\n\ndummy_list_3w = list((df_3w['article_id'].value_counts()).index)[:12]","metadata":{"id":"yc4T0JnWf-2Z","execution":{"iopub.status.busy":"2022-02-27T20:47:13.542859Z","iopub.execute_input":"2022-02-27T20:47:13.543222Z","iopub.status.idle":"2022-02-27T20:47:14.827198Z","shell.execute_reply.started":"2022-02-27T20:47:13.543171Z","shell.execute_reply":"2022-02-27T20:47:14.82637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_2w = {}\n\nfor i,x in enumerate(zip(df_2w['customer_id'], df_2w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_2w:\n        purchase_dict_2w[cust_id] = {}\n    \n    if art_id not in purchase_dict_2w[cust_id]:\n        purchase_dict_2w[cust_id][art_id] = 0\n    \n    purchase_dict_2w[cust_id][art_id] += 1\n\ndummy_list_2w = list((df_2w['article_id'].value_counts()).index)[:12]","metadata":{"id":"xuxrxnnrgE4M","execution":{"iopub.status.busy":"2022-02-27T20:47:14.829717Z","iopub.execute_input":"2022-02-27T20:47:14.830114Z","iopub.status.idle":"2022-02-27T20:47:15.730216Z","shell.execute_reply.started":"2022-02-27T20:47:14.83008Z","shell.execute_reply":"2022-02-27T20:47:15.728926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_1w = {}\n\nfor i,x in enumerate(zip(df_1w['customer_id'], df_1w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_1w:\n        purchase_dict_1w[cust_id] = {}\n    \n    if art_id not in purchase_dict_1w[cust_id]:\n        purchase_dict_1w[cust_id][art_id] = 0\n    \n    purchase_dict_1w[cust_id][art_id] += 1\n\ndummy_list_1w = list((df_1w['article_id'].value_counts()).index)[:12]","metadata":{"id":"maxlwdZIgNEJ","execution":{"iopub.status.busy":"2022-02-27T20:47:15.731917Z","iopub.execute_input":"2022-02-27T20:47:15.73223Z","iopub.status.idle":"2022-02-27T20:47:16.139095Z","shell.execute_reply.started":"2022-02-27T20:47:15.732193Z","shell.execute_reply":"2022-02-27T20:47:16.137912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_candidates(customers_id, n_candidates = 12):\n  \"\"\"\n  df - basically, dataframe with customers(customers should be unique)\n  \"\"\"\n  prediction_dict = {}\n  dummy_list = list((df_2w['article_id'].value_counts()).index)[:n_candidates]\n\n  for i, cust_id in tqdm(enumerate(customers_id)):\n    # comment this for validation\n    if cust_id in purchase_dict_1w:\n        l = sorted((purchase_dict_1w[cust_id]).items(), key=lambda x: x[1], reverse=True)\n        l = [y[0] for y in l]\n        if len(l)>n_candidates:\n            s = l[:n_candidates]\n        else:\n            s = l+dummy_list_1w[:(n_candidates-len(l))]\n    elif cust_id in purchase_dict_2w:\n        l = sorted((purchase_dict_2w[cust_id]).items(), key=lambda x: x[1], reverse=True)\n        l = [y[0] for y in l]\n        if len(l)>n_candidates:\n            s = l[:n_candidates]\n        else:\n            s = l+dummy_list_2w[:(n_candidates-len(l))]\n    elif cust_id in purchase_dict_3w:\n        l = sorted((purchase_dict_3w[cust_id]).items(), key=lambda x: x[1], reverse=True)\n        l = [y[0] for y in l]\n        if len(l)>n_candidates:\n            s = l[:n_candidates]\n        else:\n            s = l+dummy_list_3w[:(n_candidates-len(l))]\n    elif cust_id in purchase_dict_4w:\n        l = sorted((purchase_dict_4w[cust_id]).items(), key=lambda x: x[1], reverse=True)\n        l = [y[0] for y in l]\n        if len(l)>n_candidates:\n            s = l[:n_candidates]\n        else:\n            s = l+dummy_list_4w[:(n_candidates-len(l))]\n    else:\n        s = dummy_list\n    prediction_dict[cust_id] = s\n\n  k = list(map(lambda x: x[0], prediction_dict.items()))\n  v = list(map(lambda x: x[1], prediction_dict.items()))\n  negatives_df = pd.DataFrame({'customer_id': k, 'negatives': v})\n  negatives_df = (\n      negatives_df\n      .explode('negatives')\n      .rename(columns = {'negatives': 'article_id'})\n  )\n  return negatives_df","metadata":{"id":"YQaMpMKKghRG","execution":{"iopub.status.busy":"2022-02-27T20:47:16.140856Z","iopub.execute_input":"2022-02-27T20:47:16.141109Z","iopub.status.idle":"2022-02-27T20:47:16.158694Z","shell.execute_reply.started":"2022-02-27T20:47:16.141078Z","shell.execute_reply":"2022-02-27T20:47:16.157441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train model","metadata":{"id":"67_UFufELEoZ"}},{"cell_type":"code","source":"#take only last 15 transactions\ntrain['rank'] = range(len(train))\ntrain = (\n    train\n    .assign(\n        rn = train.groupby(['customer_id'])['rank']\n                  .rank(method='first', ascending=False))\n    .query(\"rn <= 15\")\n    .drop(columns = ['price', 'sales_channel_id'])\n    .sort_values(['t_dat', 'customer_id'])\n)\ntrain['label'] = 1\n\ndel train['rank']\ndel train['rn']\n\nvalid.sort_values(['t_dat', 'customer_id'], inplace = True)","metadata":{"id":"BvwlJ4qqkOxK","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Append negatives to positives using last dates from train","metadata":{"id":"gnKSsWUvt7iR"}},{"cell_type":"code","source":"last_dates = (\n    train\n    .groupby('customer_id')['t_dat']\n    .max()\n    .to_dict()\n)\n\nnegatives = prepare_candidates(train['customer_id'].unique(), 15)\nnegatives['t_dat'] = negatives['customer_id'].map(last_dates)\n\nnegatives = (\n    negatives\n    .merge(user_features, on = ('customer_id'))\n    .merge(item_features, on = ('article_id'))\n)\nnegatives['label'] = 0","metadata":{"id":"UG4znc_dyLAZ","outputId":"b41aaa1c-862a-4322-87d4-805e48ca4abc","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([train, negatives])\ntrain.sort_values(['customer_id', 't_dat'], inplace = True)","metadata":{"id":"YGHNpUPiC6lg","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_baskets = train.groupby(['customer_id'])['article_id'].count().values\n# valid_baskets = valid.groupby(['customer_id'])['article_id'].count().values\ntrain_baskets = train.groupby(['customer_id'])['article_id'].count().values","metadata":{"id":"ALBMwMkhJbdz","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fit lightgbm ranker model","metadata":{}},{"cell_type":"code","source":"ranker = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    boosting_type=\"dart\",\n    max_depth=7,\n    n_estimators=300,\n    importance_type='gain',\n    verbose=10\n)","metadata":{"id":"QlhmAP7NJbYu","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ranker = ranker.fit(\n    train.drop(columns = ['t_dat', 'customer_id', 'article_id', 'label']),\n    train.pop('label'),\n    group=train_baskets,\n)","metadata":{"id":"kQViiprFJbbM","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predictions","metadata":{"id":"LaluyLlbJJJh"}},{"cell_type":"code","source":"sample_sub = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')","metadata":{"id":"x-y7Xym7MPvy","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"candidates = prepare_candidates(sample_sub.customer_id.unique(), 12)\ncandidates = (\n    candidates\n    .merge(user_features, on = ('customer_id'))\n    .merge(item_features, on = ('article_id'))\n)","metadata":{"id":"trc0JD7BVTBA","outputId":"ab4dcff3-9061-4d35-991e-e4182022aa51","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Predict using batches, otherwise doesn't fit into memory.","metadata":{}},{"cell_type":"code","source":"preds = []\nbatch_size = 1_000_000\nfor bucket in tqdm(range(0, len(candidates), batch_size)):\n  outputs = ranker.predict(\n      candidates.iloc[bucket: bucket+batch_size]\n      .drop(columns = ['customer_id', 'article_id'])\n      )\n  preds.append(outputs)","metadata":{"id":"e9gndzGxV8Ld","outputId":"9739426b-d7ff-4c56-c73a-2e2d91602bad","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = np.concatenate(preds)\ncandidates['preds'] = preds\npreds = candidates[['customer_id', 'article_id', 'preds']]\npreds.sort_values(['customer_id', 'preds'], ascending=False, inplace = True)\npreds = (\n    preds\n    .groupby('customer_id')[['article_id']]\n    .aggregate(lambda x: x.tolist())\n)\npreds['article_id'] = preds['article_id'].apply(lambda x: ' '.join(['0'+str(k) for k in x]))","metadata":{"id":"PoIe_dn5J7mO","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Join with sample submission and fillna with articles from dummy_list_2w","metadata":{}},{"cell_type":"code","source":"preds = sample_sub[['customer_id']].merge(\n    preds\n    .reset_index()\n    .rename(columns = {'article_id': 'prediction'}), how = 'left')\npreds['prediction'].fillna(' '.join(['0'+str(art) for art in dummy_list_2w]), inplace = True)","metadata":{"id":"LQffjz8_LG5q","execution":{"iopub.status.busy":"2022-02-27T20:15:29.003726Z","iopub.execute_input":"2022-02-27T20:15:29.003968Z","iopub.status.idle":"2022-02-27T20:15:30.722655Z","shell.execute_reply.started":"2022-02-27T20:15:29.003943Z","shell.execute_reply":"2022-02-27T20:15:30.722042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.to_csv('submisssion_ranking.csv', index = False)","metadata":{"id":"DiDaFSGkMJds","execution":{"iopub.status.busy":"2022-02-27T20:15:33.157519Z","iopub.execute_input":"2022-02-27T20:15:33.157997Z","iopub.status.idle":"2022-02-27T20:15:38.888246Z","shell.execute_reply.started":"2022-02-27T20:15:33.157967Z","shell.execute_reply":"2022-02-27T20:15:38.886818Z"},"trusted":true},"execution_count":null,"outputs":[]}]}