{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook provides ranking baseline that uses item, user features and lightgbm as the ranker model. Code for preparing item features [this](https://www.kaggle.com/alexvishnevskiy/ranking-item-features), code for preparing user features [this](https://www.kaggle.com/alexvishnevskiy/ranking-user-features). Some code is taken from [this repo](https://github.com/radekosmulski/personalized_fashion_recs).\n\n\nこのノートでは、アイテム、ユーザ特徴量、lightgbmをランカーモデルとしたランキングのベースラインを提供しています。アイテム特徴量を用意するコード [こちら](https://www.kaggle.com/alexvishnevskiy/ranking-item-features)、ユーザー特徴量を用意するコード [こちら](https://www.kaggle.com/alexvishnevskiy/ranking-user-features)があります。一部のコードは [このレポ](https://github.com/radekosmulski/personalized_fashion_recs) から引用しています。","metadata":{"id":"70bkjrfStDfM"}},{"cell_type":"code","source":"from lightgbm.sklearn import LGBMRanker\nfrom datetime import timedelta\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nfrom tqdm import tqdm","metadata":{"id":"bU6vwaTcDBVB","execution":{"iopub.status.busy":"2022-04-19T02:20:56.664930Z","iopub.execute_input":"2022-04-19T02:20:56.665537Z","iopub.status.idle":"2022-04-19T02:20:57.760959Z","shell.execute_reply.started":"2022-04-19T02:20:56.665449Z","shell.execute_reply":"2022-04-19T02:20:57.760111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load all data","metadata":{}},{"cell_type":"code","source":"user_features = pd.read_parquet('../input/ranking-features/user_features.parquet')\nitem_features = pd.read_parquet('../input/ranking-features/item_features.parquet')\ntransactions_df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ntransactions_df.t_dat = pd.to_datetime( transactions_df.t_dat )","metadata":{"id":"wcdV4WcbDP_e","execution":{"iopub.status.busy":"2022-04-19T02:20:57.762545Z","iopub.execute_input":"2022-04-19T02:20:57.762765Z","iopub.status.idle":"2022-04-19T02:22:19.225305Z","shell.execute_reply.started":"2022-04-19T02:20:57.762736Z","shell.execute_reply":"2022-04-19T02:22:19.222831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Last 4 weeks of transactions will be used as a baseline.\n\n過去4週間のトランザクションをベースラインとして使用します。","metadata":{"id":"hckIzN4qswV2"}},{"cell_type":"code","source":"user_features.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:22:19.228340Z","iopub.execute_input":"2022-04-19T02:22:19.228587Z","iopub.status.idle":"2022-04-19T02:22:19.266905Z","shell.execute_reply.started":"2022-04-19T02:22:19.228564Z","shell.execute_reply":"2022-04-19T02:22:19.266484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_features.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:22:19.268744Z","iopub.execute_input":"2022-04-19T02:22:19.269106Z","iopub.status.idle":"2022-04-19T02:22:19.290077Z","shell.execute_reply.started":"2022-04-19T02:22:19.269082Z","shell.execute_reply":"2022-04-19T02:22:19.289214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_4w = transactions_df[transactions_df['t_dat'] >= pd.to_datetime('2020-08-24')].copy()\ndf_3w = transactions_df[transactions_df['t_dat'] >= pd.to_datetime('2020-08-31')].copy()\ndf_2w = transactions_df[transactions_df['t_dat'] >= pd.to_datetime('2020-09-07')].copy()\ndf_1w = transactions_df[transactions_df['t_dat'] >= pd.to_datetime('2020-09-15')].copy()","metadata":{"id":"IYyTDVwLdaTr","execution":{"iopub.status.busy":"2022-04-19T02:22:19.291812Z","iopub.execute_input":"2022-04-19T02:22:19.292143Z","iopub.status.idle":"2022-04-19T02:22:19.883703Z","shell.execute_reply.started":"2022-04-19T02:22:19.292116Z","shell.execute_reply":"2022-04-19T02:22:19.882787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Factorize all categorical features\n\nすべてのカテゴリ特徴量を因数分解(数値情報に置き換える)する","metadata":{"id":"p1OGJZ4us24-"}},{"cell_type":"code","source":"user_features[['club_member_status', 'fashion_news_frequency']]","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:22:19.884862Z","iopub.execute_input":"2022-04-19T02:22:19.885043Z","iopub.status.idle":"2022-04-19T02:22:19.914397Z","shell.execute_reply.started":"2022-04-19T02:22:19.885020Z","shell.execute_reply":"2022-04-19T02:22:19.913930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_features[['club_member_status', 'fashion_news_frequency']] = (\n                   user_features[['club_member_status', 'fashion_news_frequency']]\n                   .apply(lambda x: pd.factorize(x)[0])\n).astype('int8')","metadata":{"id":"ZHnAsBk1HNNu","execution":{"iopub.status.busy":"2022-04-19T02:22:19.915353Z","iopub.execute_input":"2022-04-19T02:22:19.916229Z","iopub.status.idle":"2022-04-19T02:22:20.150226Z","shell.execute_reply.started":"2022-04-19T02:22:19.916205Z","shell.execute_reply":"2022-04-19T02:22:20.149321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Merge user, item features to transactions.","metadata":{"id":"hLGSgcccs9zr"}},{"cell_type":"code","source":"transactions_df = (\n    transactions_df\n    .merge(user_features, on = ('customer_id'))\n    .merge(item_features, on = ('article_id'))\n)\ntransactions_df.sort_values(['t_dat', 'customer_id'], inplace=True)","metadata":{"id":"Jf0_LM-JHrUR","execution":{"iopub.status.busy":"2022-04-19T02:22:20.151500Z","iopub.execute_input":"2022-04-19T02:22:20.151660Z","iopub.status.idle":"2022-04-19T02:25:24.192278Z","shell.execute_reply.started":"2022-04-19T02:22:20.151640Z","shell.execute_reply":"2022-04-19T02:25:24.191619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:25:24.193438Z","iopub.execute_input":"2022-04-19T02:25:24.194130Z","iopub.status.idle":"2022-04-19T02:25:24.226870Z","shell.execute_reply.started":"2022-04-19T02:25:24.194093Z","shell.execute_reply":"2022-04-19T02:25:24.224623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for simplicity let's take only 1M rows\nN_ROWS = 1_000_000\n\ntrain = transactions_df.loc[ transactions_df.t_dat <= pd.to_datetime('2020-09-15') ].iloc[:N_ROWS]\nvalid = transactions_df.loc[ transactions_df.t_dat >= pd.to_datetime('2020-09-16') ]","metadata":{"id":"anyNkYukDQB1","execution":{"iopub.status.busy":"2022-04-19T02:25:24.231322Z","iopub.execute_input":"2022-04-19T02:25:24.231628Z","iopub.status.idle":"2022-04-19T02:25:35.381022Z","shell.execute_reply.started":"2022-04-19T02:25:24.231598Z","shell.execute_reply":"2022-04-19T02:25:35.379864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#delete transactions to save memory\ndel transactions_df","metadata":{"id":"B--Vk-g3dvfc","execution":{"iopub.status.busy":"2022-04-19T02:25:35.382214Z","iopub.execute_input":"2022-04-19T02:25:35.382405Z","iopub.status.idle":"2022-04-19T02:25:35.547838Z","shell.execute_reply.started":"2022-04-19T02:25:35.382382Z","shell.execute_reply":"2022-04-19T02:25:35.547126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, valid.shape","metadata":{"id":"so86O-mCE4M_","outputId":"c9fdad26-c63d-452e-cd7f-ad17d7fd44b6","execution":{"iopub.status.busy":"2022-04-19T02:25:35.549133Z","iopub.execute_input":"2022-04-19T02:25:35.549404Z","iopub.status.idle":"2022-04-19T02:25:35.564505Z","shell.execute_reply.started":"2022-04-19T02:25:35.549345Z","shell.execute_reply":"2022-04-19T02:25:35.563871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare candidates","metadata":{"id":"iQpAr_iNfkNi"}},{"cell_type":"code","source":"purchase_dict_4w = {}\n\nfor i,x in enumerate(zip(df_4w['customer_id'], df_4w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_4w:\n        purchase_dict_4w[cust_id] = {}\n    \n    if art_id not in purchase_dict_4w[cust_id]:\n        purchase_dict_4w[cust_id][art_id] = 0\n    \n    purchase_dict_4w[cust_id][art_id] += 1\n\ndummy_list_4w = list((df_4w['article_id'].value_counts()).index)[:12]","metadata":{"id":"b12XmH1SfoPR","execution":{"iopub.status.busy":"2022-04-19T02:25:35.565335Z","iopub.execute_input":"2022-04-19T02:25:35.565505Z","iopub.status.idle":"2022-04-19T02:25:37.565201Z","shell.execute_reply.started":"2022-04-19T02:25:35.565484Z","shell.execute_reply":"2022-04-19T02:25:37.564198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"週ごとに誰が、何回同じ商品を買ったのかチェック\n\n以下のような形式で保存される\n\n> `{'顧客ID(誰が？)': {商品ID(何を？): 購入回数(何回？)}}`\n","metadata":{}},{"cell_type":"code","source":"#検証用アルゴリズム\nnames = ['Alice', 'Bob', 'Charlie','Alice']\nages = [24, 50, 18,24]\ntest_dict = {}\n\nfor i, (name, age) in enumerate(zip(names, ages)):\n    print(i, name, age)\n    if name not in test_dict:\n        test_dict[name] = {}\n    \n    if age not in test_dict[name]:\n        test_dict[name][age] = 0\n    \n    test_dict[name][age] += 1\ntest_dict","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:25:37.566634Z","iopub.execute_input":"2022-04-19T02:25:37.566901Z","iopub.status.idle":"2022-04-19T02:25:37.581011Z","shell.execute_reply.started":"2022-04-19T02:25:37.566860Z","shell.execute_reply":"2022-04-19T02:25:37.580086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_3w = {}\n\nfor i,x in enumerate(zip(df_3w['customer_id'], df_3w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_3w:\n        purchase_dict_3w[cust_id] = {}\n    \n    if art_id not in purchase_dict_3w[cust_id]:\n        purchase_dict_3w[cust_id][art_id] = 0\n    \n    purchase_dict_3w[cust_id][art_id] += 1\n\ndummy_list_3w = list((df_3w['article_id'].value_counts()).index)[:12]","metadata":{"id":"yc4T0JnWf-2Z","execution":{"iopub.status.busy":"2022-04-19T02:25:37.582560Z","iopub.execute_input":"2022-04-19T02:25:37.582765Z","iopub.status.idle":"2022-04-19T02:25:39.037712Z","shell.execute_reply.started":"2022-04-19T02:25:37.582738Z","shell.execute_reply":"2022-04-19T02:25:39.036444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_2w = {}\n\nfor i,x in enumerate(zip(df_2w['customer_id'], df_2w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_2w:\n        purchase_dict_2w[cust_id] = {}\n    \n    if art_id not in purchase_dict_2w[cust_id]:\n        purchase_dict_2w[cust_id][art_id] = 0\n    \n    purchase_dict_2w[cust_id][art_id] += 1\n\ndummy_list_2w = list((df_2w['article_id'].value_counts()).index)[:12]","metadata":{"id":"xuxrxnnrgE4M","execution":{"iopub.status.busy":"2022-04-19T02:25:39.039127Z","iopub.execute_input":"2022-04-19T02:25:39.039421Z","iopub.status.idle":"2022-04-19T02:25:40.021886Z","shell.execute_reply.started":"2022-04-19T02:25:39.039388Z","shell.execute_reply":"2022-04-19T02:25:40.020835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_dict_1w = {}\n\nfor i,x in enumerate(zip(df_1w['customer_id'], df_1w['article_id'])):\n    cust_id, art_id = x\n    if cust_id not in purchase_dict_1w:\n        purchase_dict_1w[cust_id] = {}\n    \n    if art_id not in purchase_dict_1w[cust_id]:\n        purchase_dict_1w[cust_id][art_id] = 0\n    \n    purchase_dict_1w[cust_id][art_id] += 1\n\ndummy_list_1w = list((df_1w['article_id'].value_counts()).index)[:12]","metadata":{"id":"maxlwdZIgNEJ","execution":{"iopub.status.busy":"2022-04-19T02:25:40.023247Z","iopub.execute_input":"2022-04-19T02:25:40.023543Z","iopub.status.idle":"2022-04-19T02:25:40.489282Z","shell.execute_reply.started":"2022-04-19T02:25:40.023511Z","shell.execute_reply":"2022-04-19T02:25:40.488207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"prepare_candidatesでやっていること\n\n- 顧客が特定の週で最も購入している商品(特定顧客ベース)上位12をトレーニングデータに設定\n- 12の商品がなかった場合、特定の週で最も購入された商品(特定の週の全取引情報ベース)上位12のデータで不足を保管","metadata":{}},{"cell_type":"code","source":"def prepare_candidates(customers_id, n_candidates = 12):\n  \"\"\"\n  df - basically, dataframe with customers(customers should be unique)\n  \"\"\"\n  prediction_dict = {}\n  dummy_list = list((df_2w['article_id'].value_counts()).index)[:n_candidates]\n\n  for i, cust_id in tqdm(enumerate(customers_id)):\n    # comment this for validation\n    if cust_id in purchase_dict_1w:\n        # 顧客が購入したアイテムの回数のデータを参照して、降順に並び替える\n        l = sorted((purchase_dict_1w[cust_id]).items(), key=lambda x: x[1], reverse=True)\n        # 降順に並び替えたリストから、アイテムIDを配列で取得\n        l = [y[0] for y in l]\n        # 予測アイテム数の上限よりもアイテムID数が多かった場合、予測アイテム数の上限までのアイテムIDのリスト要素を取得\n        if len(l)>n_candidates:\n            s = l[:n_candidates]\n            # 予測アイテム数の上限よりもアイテムID数が少なかった場合、ダミーの値で保管\n            # ダミーの値の中身は、その週に最も購入された上位12の商品\n        else:\n            s = l+dummy_list_1w[:(n_candidates-len(l))]\n    elif cust_id in purchase_dict_2w:\n        l = sorted((purchase_dict_2w[cust_id]).items(), key=lambda x: x[1], reverse=True)\n        l = [y[0] for y in l]\n        if len(l)>n_candidates:\n            s = l[:n_candidates]\n        else:\n            s = l+dummy_list_2w[:(n_candidates-len(l))]\n    elif cust_id in purchase_dict_3w:\n        l = sorted((purchase_dict_3w[cust_id]).items(), key=lambda x: x[1], reverse=True)\n        l = [y[0] for y in l]\n        if len(l)>n_candidates:\n            s = l[:n_candidates]\n        else:\n            s = l+dummy_list_3w[:(n_candidates-len(l))]\n    elif cust_id in purchase_dict_4w:\n        l = sorted((purchase_dict_4w[cust_id]).items(), key=lambda x: x[1], reverse=True)\n        l = [y[0] for y in l]\n        if len(l)>n_candidates:\n            s = l[:n_candidates]\n        else:\n            s = l+dummy_list_4w[:(n_candidates-len(l))]\n    else:\n        s = dummy_list\n    prediction_dict[cust_id] = s\n\n  k = list(map(lambda x: x[0], prediction_dict.items()))\n  v = list(map(lambda x: x[1], prediction_dict.items()))\n  negatives_df = pd.DataFrame({'customer_id': k, 'negatives': v})\n  negatives_df = (\n      negatives_df\n      .explode('negatives')\n      .rename(columns = {'negatives': 'article_id'})\n  )\n  return negatives_df","metadata":{"id":"YQaMpMKKghRG","execution":{"iopub.status.busy":"2022-04-19T02:25:40.490599Z","iopub.execute_input":"2022-04-19T02:25:40.490813Z","iopub.status.idle":"2022-04-19T02:25:40.510928Z","shell.execute_reply.started":"2022-04-19T02:25:40.490784Z","shell.execute_reply":"2022-04-19T02:25:40.510116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train model","metadata":{"id":"67_UFufELEoZ"}},{"cell_type":"code","source":"train['rank'] = range(len(train))\ntrain.assign(rn = train.groupby(['customer_id'])['rank'].rank(method='first', ascending=False))","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:25:40.512488Z","iopub.execute_input":"2022-04-19T02:25:40.512882Z","iopub.status.idle":"2022-04-19T02:25:41.389458Z","shell.execute_reply.started":"2022-04-19T02:25:40.512843Z","shell.execute_reply":"2022-04-19T02:25:41.388634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#take only last 15 transactions\n#トレーニングデータの長さ分の数値を格納\ntrain['rank'] = range(len(train))\n#カスタマーごとに最新の15のトランザクションをトレーニングデータとして扱う\ntrain = (\n    train\n    .assign(\n        rn = train.groupby(['customer_id'])['rank']\n                  .rank(method='first', ascending=False))\n    .query(\"rn <= 15\")\n    .drop(columns = ['price', 'sales_channel_id'])\n    .sort_values(['t_dat', 'customer_id'])\n)\ntrain['label'] = 1\n\ndel train['rank']\ndel train['rn']\n\nvalid.sort_values(['t_dat', 'customer_id'], inplace = True)","metadata":{"id":"BvwlJ4qqkOxK","execution":{"iopub.status.busy":"2022-04-19T02:25:41.391149Z","iopub.execute_input":"2022-04-19T02:25:41.391462Z","iopub.status.idle":"2022-04-19T02:25:44.106369Z","shell.execute_reply.started":"2022-04-19T02:25:41.391423Z","shell.execute_reply":"2022-04-19T02:25:44.105458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Append negatives to positives using last dates from train","metadata":{"id":"gnKSsWUvt7iR"}},{"cell_type":"code","source":"#カスタマーごとに最新の購入日を取得\nlast_dates = (\n    train\n    .groupby('customer_id')['t_dat']\n    .max()\n    .to_dict()\n)\n\n# \nnegatives = prepare_candidates(train['customer_id'].unique(), 15)\nnegatives['t_dat'] = negatives['customer_id'].map(last_dates)\n\nnegatives = (\n    negatives\n    .merge(user_features, on = ('customer_id'))\n    .merge(item_features, on = ('article_id'))\n)\nnegatives['label'] = 0","metadata":{"id":"UG4znc_dyLAZ","outputId":"b41aaa1c-862a-4322-87d4-805e48ca4abc","execution":{"iopub.status.busy":"2022-04-19T02:25:44.108868Z","iopub.execute_input":"2022-04-19T02:25:44.109167Z","iopub.status.idle":"2022-04-19T02:25:53.345649Z","shell.execute_reply.started":"2022-04-19T02:25:44.109128Z","shell.execute_reply":"2022-04-19T02:25:53.344543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"negatives","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:25:53.347047Z","iopub.execute_input":"2022-04-19T02:25:53.347286Z","iopub.status.idle":"2022-04-19T02:25:54.006483Z","shell.execute_reply.started":"2022-04-19T02:25:53.347236Z","shell.execute_reply":"2022-04-19T02:25:54.005217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([train, negatives])\ntrain.sort_values(['customer_id', 't_dat'], inplace = True)","metadata":{"id":"YGHNpUPiC6lg","execution":{"iopub.status.busy":"2022-04-19T02:25:54.007972Z","iopub.execute_input":"2022-04-19T02:25:54.008216Z","iopub.status.idle":"2022-04-19T02:25:58.549823Z","shell.execute_reply.started":"2022-04-19T02:25:54.008187Z","shell.execute_reply":"2022-04-19T02:25:58.548599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"LGBMRankerは、groupプロパティに「どこからどこまでの配列が一人の顧客がどの商品購入したトランザクションデータなのか」を伝える必要があるので上記で、カスタマーIDでソートして、以下の処理で各カスタマーIDがどの商品を何回購入したかの回数を取得する\n\nその回数を配列にすることにより、「どこからどこまでの配列が一人の顧客がどの商品購入したトランザクションデータなのか」のデータ形式を満たすことができる。\n","metadata":{}},{"cell_type":"code","source":"# train_baskets = train.groupby(['customer_id'])['article_id'].count().values\n# valid_baskets = valid.groupby(['customer_id'])['article_id'].count().values\ntrain_baskets = train.groupby(['customer_id'])['article_id'].count().values","metadata":{"id":"ALBMwMkhJbdz","execution":{"iopub.status.busy":"2022-04-19T02:25:58.551169Z","iopub.execute_input":"2022-04-19T02:25:58.551430Z","iopub.status.idle":"2022-04-19T02:26:00.048694Z","shell.execute_reply.started":"2022-04-19T02:25:58.551400Z","shell.execute_reply":"2022-04-19T02:26:00.047682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_baskets","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:26:00.049983Z","iopub.execute_input":"2022-04-19T02:26:00.050181Z","iopub.status.idle":"2022-04-19T02:26:00.056627Z","shell.execute_reply.started":"2022-04-19T02:26:00.050155Z","shell.execute_reply":"2022-04-19T02:26:00.055777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fit lightgbm ranker model","metadata":{}},{"cell_type":"code","source":"ranker = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    boosting_type=\"dart\",\n    max_depth=7,\n    n_estimators=300,\n    importance_type='gain',\n    verbose=10\n)","metadata":{"id":"QlhmAP7NJbYu","execution":{"iopub.status.busy":"2022-04-19T02:26:00.057887Z","iopub.execute_input":"2022-04-19T02:26:00.058153Z","iopub.status.idle":"2022-04-19T02:26:00.069373Z","shell.execute_reply.started":"2022-04-19T02:26:00.058117Z","shell.execute_reply":"2022-04-19T02:26:00.068638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ranker = ranker.fit(\n    train.drop(columns = ['t_dat', 'customer_id', 'article_id', 'label']),\n    train.pop('label'),\n    group=train_baskets,\n)","metadata":{"id":"kQViiprFJbbM","execution":{"iopub.status.busy":"2022-04-19T02:26:00.070848Z","iopub.execute_input":"2022-04-19T02:26:00.071138Z","iopub.status.idle":"2022-04-19T02:37:03.905863Z","shell.execute_reply.started":"2022-04-19T02:26:00.071109Z","shell.execute_reply":"2022-04-19T02:37:03.905098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ranker","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:37:03.909596Z","iopub.execute_input":"2022-04-19T02:37:03.910508Z","iopub.status.idle":"2022-04-19T02:37:03.919318Z","shell.execute_reply.started":"2022-04-19T02:37:03.910476Z","shell.execute_reply":"2022-04-19T02:37:03.918730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predictions","metadata":{"id":"LaluyLlbJJJh"}},{"cell_type":"code","source":"sample_sub = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')","metadata":{"id":"x-y7Xym7MPvy","execution":{"iopub.status.busy":"2022-04-19T02:37:03.920417Z","iopub.execute_input":"2022-04-19T02:37:03.921734Z","iopub.status.idle":"2022-04-19T02:37:10.559015Z","shell.execute_reply.started":"2022-04-19T02:37:03.921667Z","shell.execute_reply":"2022-04-19T02:37:10.557502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"candidates = prepare_candidates(sample_sub.customer_id.unique(), 12)\ncandidates = (\n    candidates\n    .merge(user_features, on = ('customer_id'))\n    .merge(item_features, on = ('article_id'))\n)","metadata":{"id":"trc0JD7BVTBA","outputId":"ab4dcff3-9061-4d35-991e-e4182022aa51","execution":{"iopub.status.busy":"2022-04-19T02:37:10.561081Z","iopub.execute_input":"2022-04-19T02:37:10.561344Z","iopub.status.idle":"2022-04-19T02:37:38.388429Z","shell.execute_reply.started":"2022-04-19T02:37:10.561313Z","shell.execute_reply":"2022-04-19T02:37:38.387314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Predict using batches, otherwise doesn't fit into memory.","metadata":{}},{"cell_type":"code","source":"batch_size = 1_000_000\nfor bucket in tqdm(range(0, len(candidates), batch_size)):\n    print(bucket)\n    print(batch_size)\n    print(bucket+batch_size)\n    #candidates.iloc[bucket: bucket+batch_size]","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:37:38.389903Z","iopub.execute_input":"2022-04-19T02:37:38.390152Z","iopub.status.idle":"2022-04-19T02:37:38.407460Z","shell.execute_reply.started":"2022-04-19T02:37:38.390118Z","shell.execute_reply":"2022-04-19T02:37:38.406252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nbatch_size = 1_000_000\n# 1_000_000行ごとにcandidatesを取り出し予測\n# 予測結果はpredsに格納\nfor bucket in tqdm(range(0, len(candidates), batch_size)):\n  outputs = ranker.predict(\n      candidates.iloc[bucket: bucket+batch_size]\n      .drop(columns = ['customer_id', 'article_id'])\n      )\n  preds.append(outputs)","metadata":{"id":"e9gndzGxV8Ld","outputId":"9739426b-d7ff-4c56-c73a-2e2d91602bad","execution":{"iopub.status.busy":"2022-04-19T02:37:38.408848Z","iopub.execute_input":"2022-04-19T02:37:38.409060Z","iopub.status.idle":"2022-04-19T02:38:50.088277Z","shell.execute_reply.started":"2022-04-19T02:37:38.409030Z","shell.execute_reply":"2022-04-19T02:38:50.087796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:38:50.090958Z","iopub.execute_input":"2022-04-19T02:38:50.092537Z","iopub.status.idle":"2022-04-19T02:38:50.104143Z","shell.execute_reply.started":"2022-04-19T02:38:50.092507Z","shell.execute_reply":"2022-04-19T02:38:50.102836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = np.concatenate(preds)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:38:50.105367Z","iopub.execute_input":"2022-04-19T02:38:50.105627Z","iopub.status.idle":"2022-04-19T02:38:50.144433Z","shell.execute_reply.started":"2022-04-19T02:38:50.105597Z","shell.execute_reply":"2022-04-19T02:38:50.143203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"candidates['preds'] = preds\ncandidates['preds']","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:38:50.145930Z","iopub.execute_input":"2022-04-19T02:38:50.146872Z","iopub.status.idle":"2022-04-19T02:38:50.175065Z","shell.execute_reply.started":"2022-04-19T02:38:50.146821Z","shell.execute_reply":"2022-04-19T02:38:50.174636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = candidates[['customer_id', 'article_id', 'preds']]\npreds","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:38:50.175876Z","iopub.execute_input":"2022-04-19T02:38:50.176683Z","iopub.status.idle":"2022-04-19T02:38:51.429026Z","shell.execute_reply.started":"2022-04-19T02:38:50.176656Z","shell.execute_reply":"2022-04-19T02:38:51.427779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.sort_values(['customer_id', 'preds'], ascending=False, inplace = True)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:38:51.430479Z","iopub.execute_input":"2022-04-19T02:38:51.430787Z","iopub.status.idle":"2022-04-19T02:39:09.761728Z","shell.execute_reply.started":"2022-04-19T02:38:51.430758Z","shell.execute_reply":"2022-04-19T02:39:09.760638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = (\n    preds\n    .groupby('customer_id')[['article_id']]\n    .aggregate(lambda x: x.tolist())\n)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-04-19T02:39:09.763134Z","iopub.execute_input":"2022-04-19T02:39:09.763376Z","iopub.status.idle":"2022-04-19T02:39:30.394599Z","shell.execute_reply.started":"2022-04-19T02:39:09.763347Z","shell.execute_reply":"2022-04-19T02:39:30.394037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds['article_id'] = preds['article_id'].apply(lambda x: ' '.join(['0'+str(k) for k in x]))\npreds['article_id'] ","metadata":{"id":"PoIe_dn5J7mO","execution":{"iopub.status.busy":"2022-04-19T02:39:30.395595Z","iopub.execute_input":"2022-04-19T02:39:30.396634Z","iopub.status.idle":"2022-04-19T02:39:37.140326Z","shell.execute_reply.started":"2022-04-19T02:39:30.396600Z","shell.execute_reply":"2022-04-19T02:39:37.138898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Join with sample submission and fillna with articles from dummy_list_2w","metadata":{}},{"cell_type":"code","source":"preds = sample_sub[['customer_id']].merge(\n    preds\n    .reset_index()\n    .rename(columns = {'article_id': 'prediction'}), how = 'left')\npreds['prediction'].fillna(' '.join(['0'+str(art) for art in dummy_list_2w]), inplace = True)","metadata":{"id":"LQffjz8_LG5q","execution":{"iopub.status.busy":"2022-04-19T02:39:37.141913Z","iopub.execute_input":"2022-04-19T02:39:37.142149Z","iopub.status.idle":"2022-04-19T02:39:39.415867Z","shell.execute_reply.started":"2022-04-19T02:39:37.142116Z","shell.execute_reply":"2022-04-19T02:39:39.414990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.to_csv('submisssion_ranking.csv', index = False)","metadata":{"id":"DiDaFSGkMJds","execution":{"iopub.status.busy":"2022-04-19T02:39:39.416913Z","iopub.execute_input":"2022-04-19T02:39:39.417096Z","iopub.status.idle":"2022-04-19T02:39:44.785837Z","shell.execute_reply.started":"2022-04-19T02:39:39.417070Z","shell.execute_reply":"2022-04-19T02:39:44.784840Z"},"trusted":true},"execution_count":null,"outputs":[]}]}