{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":195911389,"sourceType":"kernelVersion"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!git clone https://github.com/Wp-Zhang/H-M-Fashion-RecSys.git","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-10T09:32:26.691723Z","iopub.execute_input":"2024-09-10T09:32:26.692144Z","iopub.status.idle":"2024-09-10T09:32:29.110169Z","shell.execute_reply.started":"2024-09-10T09:32:26.692102Z","shell.execute_reply":"2024-09-10T09:32:29.107887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working/H-M-Fashion-RecSys","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:32:29.119953Z","iopub.execute_input":"2024-09-10T09:32:29.121120Z","iopub.status.idle":"2024-09-10T09:32:29.144779Z","shell.execute_reply.started":"2024-09-10T09:32:29.121035Z","shell.execute_reply":"2024-09-10T09:32:29.142644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !mkdir -p data/raw\n# !cp -r /kaggle/input/h-and-m-personalized-fashion-recommendations/*.csv data/raw","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:32:29.147151Z","iopub.execute_input":"2024-09-10T09:32:29.154892Z","iopub.status.idle":"2024-09-10T09:32:29.165750Z","shell.execute_reply.started":"2024-09-10T09:32:29.154786Z","shell.execute_reply":"2024-09-10T09:32:29.163759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/notebook439d133a79/H-M-Fashion-RecSys/data ./","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:32:29.171652Z","iopub.execute_input":"2024-09-10T09:32:29.178341Z","iopub.status.idle":"2024-09-10T09:33:16.614695Z","shell.execute_reply.started":"2024-09-10T09:32:29.178248Z","shell.execute_reply":"2024-09-10T09:33:16.613124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U lightgbm\n!pip install implicit","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:16.617018Z","iopub.execute_input":"2024-09-10T09:33:16.617464Z","iopub.status.idle":"2024-09-10T09:33:54.512521Z","shell.execute_reply.started":"2024-09-10T09:33:16.617420Z","shell.execute_reply":"2024-09-10T09:33:54.511104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom pandas.api.types import CategoricalDtype\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport lightgbm as lgb\n\nimport pickle\nfrom tqdm import tqdm\nimport gc\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:54.515003Z","iopub.execute_input":"2024-09-10T09:33:54.515474Z","iopub.status.idle":"2024-09-10T09:33:57.622700Z","shell.execute_reply.started":"2024-09-10T09:33:54.515420Z","shell.execute_reply":"2024-09-10T09:33:57.621075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nimport sys\nfrom IPython.core.interactiveshell import InteractiveShell\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:57.624615Z","iopub.execute_input":"2024-09-10T09:33:57.625393Z","iopub.status.idle":"2024-09-10T09:33:57.632250Z","shell.execute_reply.started":"2024-09-10T09:33:57.625342Z","shell.execute_reply":"2024-09-10T09:33:57.630602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from src.data import DataHelper\nfrom src.data.metrics import map_at_k, hr_at_k, recall_at_k\n\nfrom src.retrieval.rules import (\n    OrderHistory,\n    OrderHistoryDecay,\n    ItemPair,\n    UserGroupTimeHistory,\n    UserGroupSaleTrend,\n    ItemCF,\n    TimeHistory,\n    TimeHistoryDecay,\n    SaleTrend,\n    OutOfStock,\n)\nfrom src.retrieval.collector import RuleCollector\n\nfrom src.features import full_sale, week_sale, repurchase_ratio, popularity, period_sale\n\nfrom src.utils import (\n    calc_valid_date,\n    merge_week_data,\n    reduce_mem_usage,\n    calc_embd_similarity,\n)","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:57.634247Z","iopub.execute_input":"2024-09-10T09:33:57.634789Z","iopub.status.idle":"2024-09-10T09:33:57.708740Z","shell.execute_reply.started":"2024-09-10T09:33:57.634719Z","shell.execute_reply":"2024-09-10T09:33:57.707357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = Path(\"./data/\")\nmodel_dir = Path(\"./models/\")","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:57.710554Z","iopub.execute_input":"2024-09-10T09:33:57.711278Z","iopub.status.idle":"2024-09-10T09:33:57.717854Z","shell.execute_reply.started":"2024-09-10T09:33:57.711233Z","shell.execute_reply":"2024-09-10T09:33:57.716361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_WEEK_NUM = 4\nWEEK_NUM = TRAIN_WEEK_NUM + 2\n\nVERSION_NAME = \"Recall 1\"\nTEST = True # * Set as `False` when do local experiments to save time","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:57.722330Z","iopub.execute_input":"2024-09-10T09:33:57.722855Z","iopub.status.idle":"2024-09-10T09:33:57.733970Z","shell.execute_reply.started":"2024-09-10T09:33:57.722782Z","shell.execute_reply":"2024-09-10T09:33:57.732182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n# if not os.path.exists(data_dir/\"interim\"/VERSION_NAME):\n#     os.makedirs(data_dir/\"interim\"/VERSION_NAME)\n# if not os.path.exists(data_dir/\"processed\"/VERSION_NAME):\n#     os.makedirs(data_dir/\"processed\"/VERSION_NAME)","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:57.736105Z","iopub.execute_input":"2024-09-10T09:33:57.736718Z","iopub.status.idle":"2024-09-10T09:33:57.749790Z","shell.execute_reply.started":"2024-09-10T09:33:57.736638Z","shell.execute_reply":"2024-09-10T09:33:57.747774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dh = DataHelper(data_dir)","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:57.751461Z","iopub.execute_input":"2024-09-10T09:33:57.753994Z","iopub.status.idle":"2024-09-10T09:33:57.761542Z","shell.execute_reply.started":"2024-09-10T09:33:57.753925Z","shell.execute_reply":"2024-09-10T09:33:57.760229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data = dh.preprocess_data(save=True, name=\"encoded_full\") # * run only once, processed data will be saved","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:57.764153Z","iopub.execute_input":"2024-09-10T09:33:57.764590Z","iopub.status.idle":"2024-09-10T09:33:57.774028Z","shell.execute_reply.started":"2024-09-10T09:33:57.764550Z","shell.execute_reply":"2024-09-10T09:33:57.772605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = dh.load_data(name=\"encoded_full\")","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:33:57.775561Z","iopub.execute_input":"2024-09-10T09:33:57.776008Z","iopub.status.idle":"2024-09-10T09:34:00.666309Z","shell.execute_reply.started":"2024-09-10T09:33:57.775967Z","shell.execute_reply":"2024-09-10T09:34:00.664998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uid2idx = pickle.load(open(data_dir/\"index_id_map/user_id2index.pkl\", \"rb\"))\nsubmission = pd.read_csv(data_dir/\"raw\"/'sample_submission.csv')\nsubmission['customer_id'] = submission['customer_id'].map(uid2idx)","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:34:00.667995Z","iopub.execute_input":"2024-09-10T09:34:00.668458Z","iopub.status.idle":"2024-09-10T09:34:07.536068Z","shell.execute_reply.started":"2024-09-10T09:34:00.668410Z","shell.execute_reply":"2024-09-10T09:34:07.534232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Retrieval","metadata":{}},{"cell_type":"code","source":"listBin = [-1, 19, 29, 39, 49, 59, 69, 119]\ndata['user']['age_bins'] = pd.cut(data['user']['age'], listBin)","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:34:07.537863Z","iopub.execute_input":"2024-09-10T09:34:07.538349Z","iopub.status.idle":"2024-09-10T09:34:07.609708Z","shell.execute_reply.started":"2024-09-10T09:34:07.538306Z","shell.execute_reply":"2024-09-10T09:34:07.608433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# * WEEK_NUM = 0: test\n# * WEEK_NUM = 1: valid\n# * WEEK_NUM > 1: train\nfor week in range(1,WEEK_NUM):\n    print(week, \"...\")\n    # * use sliding window to generate candidates\n    if week == 0 and not TEST:\n        continue\n    trans = data[\"inter\"]\n\n    start_date, end_date = calc_valid_date(week)\n    print(f\"Week {week}: [{start_date}, {end_date})\")\n    \n    train, valid = dh.split_data(trans, start_date, end_date)\n    train = train.merge(data['user'][['customer_id','age_bins']], on='customer_id', how='left')\n\n    last_week_start = pd.to_datetime(start_date) - pd.Timedelta(days=7)\n    last_week_start = last_week_start.strftime(\"%Y-%m-%d\")\n    last_week = train.loc[train.t_dat >= last_week_start]\n    \n    last_3day_start = pd.to_datetime(start_date) - pd.Timedelta(days=3)\n    last_3day_start = last_3day_start.strftime(\"%Y-%m-%d\")\n    last_3days = train.loc[train.t_dat >= last_3day_start]\n    \n    last_14day_start = pd.to_datetime(start_date) - pd.Timedelta(days=14)\n    last_14day_start = last_14day_start.strftime(\"%Y-%m-%d\")\n    last_14days = train.loc[train.t_dat >= last_14day_start]\n    \n    last_21day_start = pd.to_datetime(start_date) - pd.Timedelta(days=21)\n    last_21day_start = last_21day_start.strftime(\"%Y-%m-%d\")\n    last_21days = train.loc[train.t_dat >= last_21day_start]\n    \n    last_28day_start = pd.to_datetime(start_date) - pd.Timedelta(days=28)\n    last_28day_start = last_28day_start.strftime(\"%Y-%m-%d\")\n    last_28days = train.loc[train.t_dat >= last_28day_start]\n\n    if week != 0:\n        customer_list = valid[\"customer_id\"].values\n    else:\n        customer_list = submission['customer_id'].values\n\n    # * ========================== Retrieval Strategies ==========================\n    print(\"Retrieval Strategies\")\n    \n    candidates = RuleCollector().collect(\n        week_num = week,\n        trans_df = trans,\n        customer_list=customer_list,\n        rules=[\n            OrderHistory(train, days=3, name='1'),\n            OrderHistory(train, days=7, name='2'),\n            OrderHistoryDecay(train, days=3, n=50, name='1'),\n            OrderHistoryDecay(train, days=7, n=50, name='2'),\n            ItemPair(OrderHistory(train, days=3).retrieve(), name='1'),\n            ItemPair(OrderHistory(train, days=7).retrieve(), name='2'),\n            ItemPair(OrderHistoryDecay(train, days=3, n=50).retrieve(), name='3'),\n            ItemPair(OrderHistoryDecay(train, days=7, n=50).retrieve(), name='4'),\n            UserGroupTimeHistory(data, customer_list, last_week, ['age_bins'], n=50, name='1'),\n            UserGroupTimeHistory(data, customer_list, last_3days, ['age_bins'], n=50, name='2'),\n            UserGroupSaleTrend(data, customer_list, train, ['age_bins'], days=7, n=50),\n            ItemCF(last_14days, last_week, name='1'),\n            ItemCF(last_21days, last_week, name='2'),\n            ItemCF(last_28days, last_week, name='3'),\n            TimeHistory(customer_list, last_week, n=50, name='1'),\n            TimeHistory(customer_list, last_3days, n=50, name='2'),\n            TimeHistoryDecay(customer_list, train, days=3, n=50, name='1'),\n            TimeHistoryDecay(customer_list, train, days=7, n=50, name='2'),\n            SaleTrend(customer_list, train, days=7, n=50),\n            \n        ],\n        filters=[OutOfStock(trans)],\n        min_pos_rate=0.006,\n        compress=False,\n    )\n\n    candidates = (\n        pd.pivot_table(\n            candidates,\n            values=\"score\",\n            index=[\"customer_id\", \"article_id\"],\n            columns=[\"method\"],\n            aggfunc=np.sum,\n        )\n        .reset_index()\n    )\n\n    candidates.to_parquet(data_dir/\"interim\"/VERSION_NAME/f\"week{week}_candidate.pqt\")\n    valid.to_parquet(data_dir/\"processed\"/VERSION_NAME/f\"week{week}_label.pqt\")\n    ","metadata":{"execution":{"iopub.status.busy":"2024-09-10T09:34:07.611994Z","iopub.execute_input":"2024-09-10T09:34:07.612386Z","iopub.status.idle":"2024-09-10T10:55:40.592542Z","shell.execute_reply.started":"2024-09-10T09:34:07.612346Z","shell.execute_reply":"2024-09-10T10:55:40.589085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# * use the threshold in week 1 to generate candidates for test data, see the log in the upper cell \nif TEST:\n    week = 0\n    trans = data[\"inter\"]\n    \n    start_date, end_date = calc_valid_date(week)\n    print(f\"Week {week}: [{start_date}, {end_date})\")\n    \n    train, valid = dh.split_data(trans, start_date, end_date)\n    train = train.merge(data['user'][['customer_id','age_bins']], on='customer_id', how='left')\n\n    last_week_start = pd.to_datetime(start_date) - pd.Timedelta(days=7)\n    last_week_start = last_week_start.strftime(\"%Y-%m-%d\")\n    last_week = train.loc[train.t_dat >= last_week_start]\n    \n    last_3day_start = pd.to_datetime(start_date) - pd.Timedelta(days=3)\n    last_3day_start = last_3day_start.strftime(\"%Y-%m-%d\")\n    last_3days = train.loc[train.t_dat >= last_3day_start]\n    \n    last_14day_start = pd.to_datetime(start_date) - pd.Timedelta(days=14)\n    last_14day_start = last_14day_start.strftime(\"%Y-%m-%d\")\n    last_14days = train.loc[train.t_dat >= last_14day_start]\n    \n    last_21day_start = pd.to_datetime(start_date) - pd.Timedelta(days=21)\n    last_21day_start = last_21day_start.strftime(\"%Y-%m-%d\")\n    last_21days = train.loc[train.t_dat >= last_21day_start]\n    \n    last_28day_start = pd.to_datetime(start_date) - pd.Timedelta(days=28)\n    last_28day_start = last_28day_start.strftime(\"%Y-%m-%d\")\n    last_28days = train.loc[train.t_dat >= last_28day_start]\n\n    customer_list = submission['customer_id'].values\n\n    # * ========================== Retrieval Strategies ==========================\n\n    candidates = RuleCollector().collect(\n        week_num = week,\n        trans_df = trans,\n        customer_list=customer_list,\n        rules=[\n            OrderHistory(train, days=3, name='1'),\n            OrderHistory(train, days=7, name='2'),\n            OrderHistoryDecay(train, days=3, n=50, name='1'),\n            OrderHistoryDecay(train, days=7, n=50, name='2'),\n            ItemPair(OrderHistory(train, days=3).retrieve(), name='1'),\n            ItemPair(OrderHistory(train, days=7).retrieve(), name='2'),\n            ItemPair(OrderHistoryDecay(train, 3, n=50).retrieve(), name='3'),\n            ItemPair(OrderHistoryDecay(train, 7, n=50).retrieve(), name='4'),\n            UserGroupTimeHistory(data, customer_list, last_week, ['age_bins'], n=15, name='1'),\n            UserGroupTimeHistory(data, customer_list, last_3days, ['age_bins'], n=20.5, name='2'),\n            UserGroupSaleTrend(data, customer_list, train, ['age_bins'], days=7, n=2),\n            ItemCF(last_14days, last_week, name='1'),\n            ItemCF(last_21days, last_week, name='2'),\n            ItemCF(last_28days, last_week, name='3'),\n            TimeHistory(customer_list, last_week, n=9, name='1'),\n            TimeHistory(customer_list, last_3days, n=16, name='2'),\n            TimeHistoryDecay(customer_list, train, days=3, n=12, name='1'),\n            TimeHistoryDecay(customer_list, train, days=7, n=8, name='2'),\n            SaleTrend(customer_list, train, days=7, n=2),\n        ],\n        filters=[OutOfStock(trans)],\n        min_pos_rate=0.006,\n        compress=False,\n    )\n    \n    candidates, _ = reduce_mem_usage(candidates)\n    candidates = (\n        pd.pivot_table(\n            candidates,\n            values=\"score\",\n            index=[\"customer_id\", \"article_id\"],\n            columns=[\"method\"],\n            aggfunc=np.sum,\n        )\n        .reset_index()\n    )\n\n    candidates.to_parquet(data_dir/\"interim\"/VERSION_NAME/f\"week{week}_candidate.pqt\")\n    valid.to_parquet(data_dir/\"processed\"/VERSION_NAME/f\"week{week}_label.pqt\")","metadata":{"execution":{"iopub.status.busy":"2024-09-10T10:55:40.595207Z","iopub.status.idle":"2024-09-10T10:55:40.596010Z","shell.execute_reply.started":"2024-09-10T10:55:40.595583Z","shell.execute_reply":"2024-09-10T10:55:40.595618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, valid, last_week, customer_list, candidates\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-09-10T10:55:40.597674Z","iopub.status.idle":"2024-09-10T10:55:40.598228Z","shell.execute_reply.started":"2024-09-10T10:55:40.597982Z","shell.execute_reply":"2024-09-10T10:55:40.598007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Adding features","metadata":{}},{"cell_type":"code","source":"# user = data[\"user\"]\n# item = data[\"item\"]\n# inter = data[\"inter\"]","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:26:42.072694Z","iopub.execute_input":"2024-09-09T09:26:42.074018Z","iopub.status.idle":"2024-09-09T09:26:42.079423Z","shell.execute_reply.started":"2024-09-09T09:26:42.073954Z","shell.execute_reply":"2024-09-09T09:26:42.078241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate week number\n# inter['week'] = (pd.to_datetime('2020-09-29') - pd.to_datetime(inter['t_dat'])).dt.days // 7","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:26:42.316132Z","iopub.execute_input":"2024-09-09T09:26:42.316726Z","iopub.status.idle":"2024-09-09T09:26:50.412085Z","shell.execute_reply.started":"2024-09-09T09:26:42.316672Z","shell.execute_reply":"2024-09-09T09:26:50.410699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter.sample(10)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:26:50.414786Z","iopub.execute_input":"2024-09-09T09:26:50.415447Z","iopub.status.idle":"2024-09-09T09:26:52.113391Z","shell.execute_reply.started":"2024-09-09T09:26:50.415387Z","shell.execute_reply":"2024-09-09T09:26:52.112099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # merge full candidates to transaction data (avoid feature missing in training data)\n# full_candidates = []\n# for i in tqdm(range(1, WEEK_NUM)):\n#     candidate = pd.read_parquet(data_dir/\"interim\"/VERSION_NAME/f\"week{i}_candidate.pqt\")\n#     full_candidates += candidate['article_id'].values.tolist()\n# full_candidates = list(set(full_candidates))\n# del candidate\n# gc.collect()\n\n# num_candidates = len(full_candidates)\n# full_candidates = np.array(full_candidates)\n# full_candidates = np.tile(full_candidates, WEEK_NUM + 1)\n# weeks = np.repeat(np.arange(1,WEEK_NUM+2), num_candidates)\n# full_candidates = pd.DataFrame({'article_id':full_candidates, 'week':weeks})\n\n# inter['valid'] = 1\n# in_train = inter[inter['week']<=WEEK_NUM + 1]\n# out_train = inter[inter['week']>WEEK_NUM + 1]\n\n# in_train = in_train.merge(full_candidates, on=['article_id','week'], how='right')\n# in_train['valid'] = in_train['valid'].fillna(0)\n# inter = pd.concat([in_train, out_train], ignore_index=True)\n# inter = inter.sort_values([\"valid\"], ascending=False).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:26:52.11516Z","iopub.execute_input":"2024-09-09T09:26:52.115683Z","iopub.status.idle":"2024-09-09T09:27:36.476247Z","shell.execute_reply.started":"2024-09-09T09:26:52.115628Z","shell.execute_reply":"2024-09-09T09:27:36.47499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # merge `product_code`\n# inter = inter.merge(item[[\"article_id\", \"product_code\"]], on=\"article_id\", how=\"left\")","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:27:36.478875Z","iopub.execute_input":"2024-09-09T09:27:36.479249Z","iopub.status.idle":"2024-09-09T09:27:41.839921Z","shell.execute_reply.started":"2024-09-09T09:27:36.479211Z","shell.execute_reply":"2024-09-09T09:27:41.838419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# base_cols = inter.columns[:8].to_list()\n# base_inter = inter[base_cols].copy()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:27:41.84156Z","iopub.execute_input":"2024-09-09T09:27:41.842006Z","iopub.status.idle":"2024-09-09T09:27:44.961641Z","shell.execute_reply.started":"2024-09-09T09:27:41.841962Z","shell.execute_reply":"2024-09-09T09:27:44.96034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter['valid'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:27:44.963391Z","iopub.execute_input":"2024-09-09T09:27:44.963835Z","iopub.status.idle":"2024-09-09T09:27:45.503992Z","shell.execute_reply.started":"2024-09-09T09:27:44.963792Z","shell.execute_reply":"2024-09-09T09:27:45.502573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from typing import List\n\n# def period_sale(\n#     trans: pd.DataFrame,\n#     groupby_cols: List,\n#     unique=False,\n#     days: int = 14,\n#     rank: bool = False,\n#     norm: bool = False,\n#     week_num: int = 6,\n# ) -> np.array:\n#     \"\"\"Calculate item unit sale in last n days.\n\n#     Parameters\n#     ----------\n#     trans : pd.DataFrame\n#         Dataframe of transaction data.\n#     groupby_cols : List\n#         Item unit.\n#     unique : bool, optional\n#         Whether to drop duplicate customer-item pairs, by default ``False``.\n#     days : int, optional\n#         Length of time window, by default ``14``.\n#     rank: bool, optional\n#         Whether to return rank of sale, by default ``False``.\n#     norm: bool, optional\n#         Whether to normalized count, by default ``False``.\n#     week_num : int, optional\n#         Number of weeks of data to calculate sale for, by default ``6``.\n\n#     Returns\n#     -------\n#     Tuple[np.array]\n#         Period sale (sale rank | normed sale).\n#     \"\"\"\n#     df = trans[[*groupby_cols, \"customer_id\", \"t_dat\", \"week\", \"valid\"]]\n#     if unique:\n#         df = df.drop_duplicates([\"customer_id\", *groupby_cols])\n\n#     df[\"t_dat\"] = pd.to_datetime(df[\"t_dat\"])\n\n#     tmp_l = []\n#     name = \"PERIOD_SALE\"\n#     for week in range(1, week_num + 1):\n#         _, end_date = calc_valid_date(week)\n#         tmp_df = df[\n#             (pd.to_datetime(end_date) - pd.Timedelta(days=days + 1) <= df[\"t_dat\"])\n#             & (df[\"t_dat\"] < pd.to_datetime(end_date))\n#         ]\n#         sale = tmp_df.groupby(groupby_cols)[\"valid\"].sum().reset_index(name=name)\n#         sale[\"week\"] = week\n#         if rank:\n#             sale[name + \"_rank\"] = sale[name].rank(ascending=False)\n#         if norm:\n#             sale[name + \"_norm\"] = sale[name] / sale[name].sum()\n\n#         tmp_l.append(sale)\n\n#     sale_df = pd.concat(tmp_l, ignore_index=True)\n#     df = df.merge(sale_df, on=[*groupby_cols, \"week\"], how=\"left\")\n\n#     if not rank and not norm:\n#         return df[name].values.astype(np.int32)\n#     elif rank and not norm:\n#         return df[name].values.astype(np.int32), df[name + \"_rank\"].values.astype(np.int32)\n#     elif not rank and norm:\n#         return df[name].values.astype(np.int32), df[name + \"_norm\"].values.astype('float32')\n#     else:\n#         return (\n#             df[name].values.astype(np.int32),\n#             df[name + \"_rank\"].values.astype(np.int32),\n#             df[name + \"_norm\"].values.astype('float32'),\n#         )","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:27:45.506046Z","iopub.execute_input":"2024-09-09T09:27:45.506563Z","iopub.status.idle":"2024-09-09T09:27:45.525367Z","shell.execute_reply.started":"2024-09-09T09:27:45.506513Z","shell.execute_reply":"2024-09-09T09:27:45.524009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# _, inter[\"i_1w_sale_rank\"], inter[\"i_1w_sale_norm\"] = period_sale(\n#     inter, [\"article_id\"], days=14, rank=True, norm=True, week_num=WEEK_NUM\n# )\n# _, inter[\"p_1w_sale_rank\"], inter[\"p_1w_sale_norm\"] = period_sale(\n#     inter, [\"product_code\"], days=14, rank=True, norm=True, week_num=WEEK_NUM\n# )\n# inter[\"i_2w_sale\"], inter[\"i_2w_sale_rank\"], inter[\"i_2w_sale_norm\"] = period_sale(\n#     inter, [\"article_id\"], days=14, rank=True, norm=True, week_num=WEEK_NUM\n# )\n# inter[\"p_2w_sale\"], inter[\"p_2w_sale_rank\"], inter[\"p_2w_sale_norm\"] = period_sale(\n#     inter, [\"product_code\"], days=14, rank=True, norm=True, week_num=WEEK_NUM\n# )","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:27:45.529193Z","iopub.execute_input":"2024-09-09T09:27:45.529656Z","iopub.status.idle":"2024-09-09T09:28:51.665487Z","shell.execute_reply.started":"2024-09-09T09:27:45.52961Z","shell.execute_reply":"2024-09-09T09:28:51.663944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter[\"i_3w_sale\"], inter[\"i_3w_sale_rank\"], inter[\"i_3w_sale_norm\"] = period_sale(\n#     inter, [\"article_id\"], days=21, rank=True, norm=True, week_num=WEEK_NUM\n# )\n# inter[\"p_3w_sale\"], inter[\"p_3w_sale_rank\"], inter[\"p_3w_sale_norm\"] = period_sale(\n#     inter, [\"product_code\"], days=21, rank=True, norm=True, week_num=WEEK_NUM\n# )\n# inter[\"i_4w_sale\"], inter[\"i_4w_sale_rank\"], inter[\"i_4w_sale_norm\"] = period_sale(\n#     inter, [\"article_id\"], days=28, rank=True, norm=True, week_num=WEEK_NUM\n# )\n# inter[\"p_4w_sale\"], inter[\"p_4w_sale_rank\"], inter[\"p_4w_sale_norm\"] = period_sale(\n#     inter, [\"product_code\"], days=28, rank=True, norm=True, week_num=WEEK_NUM\n# )","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:28:51.667231Z","iopub.execute_input":"2024-09-09T09:28:51.667623Z","iopub.status.idle":"2024-09-09T09:29:57.107606Z","shell.execute_reply.started":"2024-09-09T09:28:51.667583Z","shell.execute_reply":"2024-09-09T09:29:57.10638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter.sample(10)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:29:57.109244Z","iopub.execute_input":"2024-09-09T09:29:57.109678Z","iopub.status.idle":"2024-09-09T09:29:58.915024Z","shell.execute_reply.started":"2024-09-09T09:29:57.109634Z","shell.execute_reply":"2024-09-09T09:29:58.913925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter['i_repurchase_ratio'] = repurchase_ratio(base_inter, ['article_id'], week_num=WEEK_NUM)\n# inter['p_repurchase_ratio'] = repurchase_ratio(base_inter, ['product_code'], week_num=WEEK_NUM)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:29:58.916612Z","iopub.execute_input":"2024-09-09T09:29:58.917073Z","iopub.status.idle":"2024-09-09T09:34:52.822169Z","shell.execute_reply.started":"2024-09-09T09:29:58.917028Z","shell.execute_reply":"2024-09-09T09:34:52.820742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter.sample(5)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:34:52.823867Z","iopub.execute_input":"2024-09-09T09:34:52.824309Z","iopub.status.idle":"2024-09-09T09:34:54.643428Z","shell.execute_reply.started":"2024-09-09T09:34:52.824265Z","shell.execute_reply":"2024-09-09T09:34:54.642208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter.shape","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:34:54.645176Z","iopub.execute_input":"2024-09-09T09:34:54.646083Z","iopub.status.idle":"2024-09-09T09:34:54.653814Z","shell.execute_reply.started":"2024-09-09T09:34:54.646024Z","shell.execute_reply":"2024-09-09T09:34:54.652503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def reduce_mem_usage(df: pd.DataFrame, verbose: bool = False) -> pd.DataFrame:\n#     \"\"\"Rudce memory usage by changing feature dtype.\n\n#     Parameters\n#     ----------\n#     df : pd.DataFrame\n#         Dataframe to reduce memory usage.\n#     verbose : bool, optional\n#         Whether to print the process, by defaults ``False``.\n\n#     Returns\n#     -------\n#     pd.DataFrame\n#         Reduced memory usage dataframe.\n\n#     References\n#     ----------\n#     .. [1] https://www.kaggle.com/arjanso/reducing-dataframe-memory-size-by-65\n\n#     \"\"\"\n#     start_mem_usg = df.memory_usage().sum() / 1024**2\n#     if verbose:\n#         print(\"Memory usage of dataframe is :\", start_mem_usg, \" MB\")\n#     NAlist = []  # Keeps track of columns that have missing values filled in.\n#     for col in df.columns:\n#         if df[col].dtype != object:  # Exclude strings\n\n#             # Print current column type\n#             if verbose:\n#                 print(\"******************************\")\n#                 print(\"Column: \", col)\n#                 print(\"dtype before: \", df[col].dtype)\n\n#             # make variables for Int, max and min\n#             IsInt = False\n#             mx = df[col].max()\n#             mn = df[col].min()\n\n#             # Integer does not support NA, therefore, NA needs to be filled\n#             if not np.isfinite(df[col]).all():\n#                 NAlist.append(col)\n#                 df[col].fillna(mn - 1, inplace=True)\n\n#             # test if column can be converted to an integer\n#             if pd.api.types.is_integer_dtype(df[col]):\n#                 IsInt = True\n#             else:\n#                 asint = df[col].fillna(0).astype(np.int64)\n#                 result = df[col] - asint\n#                 result = result.sum()\n#                 if result > -0.01 and result < 0.01:\n#                     IsInt = True\n\n#             # Make Integer/unsigned Integer datatypes\n#             if IsInt:\n#                 if mn >= 0:\n#                     if mx <= 255:\n#                         df[col] = df[col].astype(np.uint8)\n#                     elif mx <= 65535:\n#                         df[col] = df[col].astype(np.uint16)\n#                     elif mx <= 4294967295:\n#                         df[col] = df[col].astype(np.uint32)\n#                     else:\n#                         df[col] = df[col].astype(np.uint64)\n#                 else:\n#                     if mn >= np.iinfo(np.int8).min and mx <= np.iinfo(np.int8).max:\n#                         df[col] = df[col].astype(np.int8)\n#                     elif mn >= np.iinfo(np.int16).min and mx <= np.iinfo(np.int16).max:\n#                         df[col] = df[col].astype(np.int16)\n#                     elif mn >= np.iinfo(np.int32).min and mx <= np.iinfo(np.int32).max:\n#                         df[col] = df[col].astype(np.int32)\n#                     elif mn >= np.iinfo(np.int64).min and mx <= np.iinfo(np.int64).max:\n#                         df[col] = df[col].astype(np.int64)\n\n#             # Make float datatypes 32 bit\n#             else:\n#                 df[col] = df[col].astype(np.float32)\n\n#             # Print new column type\n#             if verbose:\n#                 print(\"dtype after: \", df[col].dtype)\n#                 print(\"******************************\")\n\n#     # Print final result\n#     mem_usg = df.memory_usage().sum() / 1024**2\n#     if verbose:\n#         print(\"___MEMORY USAGE AFTER COMPLETION:___\")\n#         print(\"Memory usage is: \", mem_usg, \" MB\")\n#         print(\"This is \", 100 * mem_usg / start_mem_usg, \"% of the initial size\")\n#     return df, NAlist\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:34:54.655342Z","iopub.execute_input":"2024-09-09T09:34:54.65588Z","iopub.status.idle":"2024-09-09T09:34:54.680226Z","shell.execute_reply.started":"2024-09-09T09:34:54.65576Z","shell.execute_reply":"2024-09-09T09:34:54.678852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter, _ = reduce_mem_usage(inter)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:34:54.682081Z","iopub.execute_input":"2024-09-09T09:34:54.682537Z","iopub.status.idle":"2024-09-09T09:35:13.11868Z","shell.execute_reply.started":"2024-09-09T09:34:54.682492Z","shell.execute_reply":"2024-09-09T09:35:13.117406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# base_inter, _ = reduce_mem_usage(base_inter)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:35:13.120245Z","iopub.execute_input":"2024-09-09T09:35:13.120678Z","iopub.status.idle":"2024-09-09T09:35:17.759877Z","shell.execute_reply.started":"2024-09-09T09:35:13.120632Z","shell.execute_reply":"2024-09-09T09:35:17.758741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:42:09.93815Z","iopub.execute_input":"2024-09-09T09:42:09.93886Z","iopub.status.idle":"2024-09-09T09:42:10.159233Z","shell.execute_reply.started":"2024-09-09T09:42:09.938784Z","shell.execute_reply":"2024-09-09T09:42:10.157518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def week_sale(\n#     trans: pd.DataFrame,\n#     groupby_cols: List,\n#     unique=False,\n#     step: int = 0,\n#     week_num: int = 6,\n# ) -> np.ndarray:\n#     \"\"\"Calculate week sales of each item unit.\n\n#     Parameters\n#     ----------\n#     trans : pd.DataFrame\n#         Dataframe of transaction data.\n#     groupby_cols : List\n#         Item unit.\n#     unique : bool, optional\n#         Whether to drop duplicate customer-item pairs, by default ``False``.\n#     step: int, optional\n#         Step of week, by default ``0``. 0 means current week sale, 1 means last week sale, etc.\n\n#     Returns\n#     -------\n#     np.ndarray\n#         Array of week sales.\n#     \"\"\"\n\n#     tmp_inter = trans[[\"week\", \"customer_id\", \"valid\", *groupby_cols]]\n#     tmp_inter = tmp_inter[tmp_inter[\"week\"] <= week_num + step + 2]\n#     if unique:\n#         tmp_inter = tmp_inter.drop_duplicates([\"customer_id\", *groupby_cols])\n\n#     df = (\n#         tmp_inter.groupby([\"week\", *groupby_cols])[\"valid\"]\n#         .sum()\n#         .reset_index(name=\"_SALE\")\n#     )\n#     df[\"week\"] -= step\n\n#     tmp_inter = trans[[\"week\", \"customer_id\", *groupby_cols]].merge(\n#         df, on=[\"week\", *groupby_cols], how=\"left\"\n#     )\n#     tmp_inter[\"_SALE\"] = tmp_inter[\"_SALE\"].fillna(0).astype('int32')\n\n#     return tmp_inter[\"_SALE\"].values\n\n# inter[\"i_sale\"] = week_sale(base_inter, [\"article_id\"], week_num=WEEK_NUM)\n# inter[\"p_sale\"] = week_sale(base_inter, [\"product_code\"], week_num=WEEK_NUM)\n# inter[\"i_sale_uni\"] = week_sale(base_inter, [\"article_id\"], True, week_num=WEEK_NUM)\n# inter[\"p_sale_uni\"] = week_sale(base_inter, [\"product_code\"], True, week_num=WEEK_NUM)\n# inter[\"lw_i_sale\"] = week_sale(base_inter, [\"article_id\"], step=1, week_num=WEEK_NUM) # * last week sale\n# inter[\"lw_p_sale\"] = week_sale(base_inter, [\"product_code\"], step=1, week_num=WEEK_NUM)\n# inter[\"lw_i_sale_uni\"] = week_sale(base_inter, [\"article_id\"], True, step=1, week_num=WEEK_NUM)\n# inter[\"lw_p_sale_uni\"] = week_sale(base_inter, [\"product_code\"], True, step=1, week_num=WEEK_NUM)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:42:11.771352Z","iopub.execute_input":"2024-09-09T09:42:11.771877Z","iopub.status.idle":"2024-09-09T09:43:15.124057Z","shell.execute_reply.started":"2024-09-09T09:42:11.771818Z","shell.execute_reply":"2024-09-09T09:43:15.122702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter[\"i_sale_ratio\"] = (inter[\"i_sale\"] / (inter[\"p_sale\"] + 1e-6)).astype(np.float32, copy=False)\n# inter[\"i_sale_uni_ratio\"] = (inter[\"i_sale_uni\"] / (inter[\"p_sale_uni\"] + 1e-6)).astype(np.float32, copy=False)\n# inter[\"lw_i_sale_ratio\"] = (inter[\"lw_i_sale\"] / (inter[\"lw_p_sale\"] + 1e-6)).astype(np.float32, copy=False)\n# inter[\"lw_i_sale_uni_ratio\"] = (inter[\"lw_i_sale_uni\"] / (inter[\"lw_p_sale_uni\"] + 1e-6)).astype(np.float32, copy=False)\n\n# inter[\"i_uni_ratio\"] = (inter[\"i_sale\"] / (inter[\"i_sale_uni\"] + 1e-6)).astype(np.float32, copy=False)\n# inter[\"p_uni_ratio\"] = (inter[\"p_sale\"] / (inter[\"p_sale_uni\"] + 1e-6)).astype(np.float32, copy=False)\n# inter[\"lw_i_uni_ratio\"] = (inter[\"lw_i_sale\"] / (inter[\"lw_i_sale_uni\"] + 1e-6)).astype(np.float32, copy=False)\n# inter[\"lw_p_uni_ratio\"] = (inter[\"lw_p_sale\"] / (inter[\"lw_p_sale_uni\"] + 1e-6)).astype(np.float32, copy=False)\n\n# inter[\"i_sale_trend\"] = ((inter[\"i_sale\"] - inter[\"lw_i_sale\"]) / (inter[\"lw_i_sale\"] + 1e-6)).astype(np.float32, copy=False)\n# inter[\"p_sale_trend\"] = ((inter[\"p_sale\"] - inter[\"lw_p_sale\"]) / (inter[\"lw_p_sale\"] + 1e-6)).astype(np.float32, copy=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:43:15.126331Z","iopub.execute_input":"2024-09-09T09:43:15.126735Z","iopub.status.idle":"2024-09-09T09:43:18.382589Z","shell.execute_reply.started":"2024-09-09T09:43:15.126694Z","shell.execute_reply":"2024-09-09T09:43:18.381415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# item_feats = [\n#     \"product_type_no\",\n#     # \"product_group_name\",\n#     # \"graphical_appearance_no\",\n#     # \"colour_group_code\",\n#     # \"perceived_colour_value_id\",\n#     # \"perceived_colour_master_id\",\n# ]\n# inter = inter.merge(item[[\"article_id\", *item_feats]], on=\"article_id\", how=\"left\")\n\n# for f in tqdm(item_feats):\n#     inter[f\"{f}_sale\"] = week_sale(inter[base_cols + [f]], [f], f\"{f}_sale\", week_num=WEEK_NUM)\n#     inter[f\"lw_{f}_sale\"] = week_sale(inter[base_cols + [f]], [f], f\"{f}_sale\", step=1, week_num=WEEK_NUM)\n#     inter[f\"{f}_sale_trend\"] = (inter[f\"{f}_sale\"] - inter[f\"lw_{f}_sale\"]) / (inter[f\"lw_{f}_sale\"] + 1e-6).astype(np.float32, copy=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:43:18.384177Z","iopub.execute_input":"2024-09-09T09:43:18.384556Z","iopub.status.idle":"2024-09-09T09:43:51.148221Z","shell.execute_reply.started":"2024-09-09T09:43:18.384512Z","shell.execute_reply":"2024-09-09T09:43:51.147047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # * Date related\n# curr_date_dict = {x:calc_valid_date(x-1)[0] for x in range(100)}\n# current_dat = inter['week'].map(curr_date_dict)\n# mask = inter['valid']==0\n# inter.loc[mask, 't_dat'] = inter.loc[mask, 'week'].map(curr_date_dict)\n# first_date = inter.groupby('article_id')['t_dat'].min().reset_index(name='first_dat')\n# inter['first_dat'] = pd.merge(inter['article_id'], first_date, on='article_id', how='left')['first_dat']\n# # df = pd.merge(df, last_date, on='article_id', how='left')\n# inter['first_dat'] = (pd.to_datetime(current_dat)-pd.to_datetime(inter['first_dat'])).dt.days","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:43:51.151182Z","iopub.execute_input":"2024-09-09T09:43:51.151578Z","iopub.status.idle":"2024-09-09T09:44:27.73308Z","shell.execute_reply.started":"2024-09-09T09:43:51.151528Z","shell.execute_reply":"2024-09-09T09:44:27.73172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def full_sale(\n#     trans: pd.DataFrame,\n#     groupby_cols: List,\n#     unique=False,\n#     week_num: int = 6,\n# ) -> np.ndarray:\n#     \"\"\"Calculate cumulative sales of each item unit.\n\n#     Parameters\n#     ----------\n#     trans : pd.DataFrame\n#         Dataframe of transaction data.\n#     groupby_cols : List\n#         Item unit.\n#     unique : bool, optional\n#         Whether to drop duplicate customer-item pairs, by default ``False``.\n\n#     Returns\n#     -------\n#     np.ndarray\n#         Array of cumulative sales.\n#     \"\"\"\n#     inter = trans[[\"customer_id\", \"week\", \"valid\", *groupby_cols]]\n#     if unique:\n#         inter = inter.drop_duplicates([\"customer_id\", \"week\", *groupby_cols])\n\n#     tmp_l = []\n#     for week in range(1, week_num + 1):\n#         df = inter[inter[\"week\"] >= week]\n#         df = df.groupby([*groupby_cols])[\"valid\"].sum().reset_index(name=\"_SALE\")\n#         df[\"week\"] = week\n#         tmp_l.append(df)\n\n#     df = pd.concat(tmp_l, ignore_index=True)\n#     inter = trans[[\"customer_id\", \"week\", *groupby_cols]].merge(\n#         df, on=[\"week\", *groupby_cols], how=\"left\"\n#     )\n#     inter[\"_SALE\"] = inter[\"_SALE\"].fillna(0).astype(\"int32\")\n\n#     return inter[\"_SALE\"].values\n\n# inter['i_full_sale'] = full_sale(base_inter, ['article_id'], week_num=WEEK_NUM)\n# inter['p_full_sale'] = full_sale(base_inter, ['product_code'], week_num=WEEK_NUM)\n\n# inter['i_daily_sale'] = (inter['i_full_sale'] / (inter['first_dat'] + 1e-6)).astype(np.float32, copy=False)\n# inter['p_daily_sale'] = (inter['p_full_sale'] / (inter['first_dat'] + 1e-6)).astype(np.float32, copy=False)\n# inter['i_daily_sale_ratio'] = (inter['i_daily_sale'] / (inter['p_daily_sale'] + 1e-6)).astype(np.float32, copy=False)\n# inter['i_w_full_sale_ratio'] = (inter['i_sale'] / (inter['i_full_sale'] + 1e-6)).astype(np.float32, copy=False)\n\n# inter['i_2w_full_sale_ratio'] = (inter['i_2w_sale'] / (inter['i_full_sale'] + 1e-6)).astype(np.float32, copy=False)\n# inter['p_w_full_sale_ratio'] = (inter['p_sale'] / (inter['p_full_sale'] + 1e-6)).astype(np.float32, copy=False)\n# inter['p_2w_full_sale_ratio'] = (inter['p_2w_sale'] / (inter['p_full_sale'] + 1e-6)).astype(np.float32, copy=False)\n\n# inter['i_week_above_daily_sale'] = (inter['i_sale'] / 7 - inter['i_daily_sale']).astype(np.float32, copy=False)\n# inter['p_week_above_full_sale'] = (inter['p_sale'] / 7 - inter['i_full_sale']).astype(np.float32, copy=False)\n# inter['i_2w_week_above_daily_sale'] = (inter['i_2w_sale'] / 14 - inter['i_daily_sale']).astype(np.float32, copy=False)\n# inter['p_2w_week_above_daily_sale'] = (inter['p_2w_sale'] / 14 - inter['p_daily_sale']).astype(np.float32, copy=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:44:27.734753Z","iopub.execute_input":"2024-09-09T09:44:27.735191Z","iopub.status.idle":"2024-09-09T09:45:10.649765Z","shell.execute_reply.started":"2024-09-09T09:44:27.735149Z","shell.execute_reply":"2024-09-09T09:45:10.648407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:45:10.651258Z","iopub.execute_input":"2024-09-09T09:45:10.651688Z","iopub.status.idle":"2024-09-09T09:45:10.805132Z","shell.execute_reply.started":"2024-09-09T09:45:10.651646Z","shell.execute_reply":"2024-09-09T09:45:10.80392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for f in tqdm(item_feats):\n#     inter[f'{f}_full_sale'] = full_sale(inter[base_cols + [f]], [f], week_num=WEEK_NUM)\n#     f_first_date = inter.groupby(f)['t_dat'].min().reset_index(name=f'{f}_first_dat')\n#     inter[f'{f}_first_dat'] = pd.merge(inter[f], f_first_date, on=f, how='left')[f'{f}_first_dat']\n#     inter[f'{f}_daily_sale'] = inter[f'{f}_full_sale'] / (pd.to_datetime(current_dat) - pd.to_datetime(inter[f'{f}_first_dat'])).dt.days.astype(np.float32, copy=False)\n#     inter[f'i_{f}_daily_sale_ratio'] = (inter['i_daily_sale'] / (inter[f'{f}_daily_sale'] + 1e-6)).astype(np.float32, copy=False)\n#     inter[f'p_{f}_daily_sale_ratio'] = (inter['p_daily_sale'] / (inter[f'{f}_daily_sale'] + 1e-6)).astype(np.float32, copy=False)\n#     del inter[f'{f}_full_sale'], inter[f'{f}_first_dat']\n#     gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:45:10.806483Z","iopub.execute_input":"2024-09-09T09:45:10.806887Z","iopub.status.idle":"2024-09-09T09:45:57.014207Z","shell.execute_reply.started":"2024-09-09T09:45:10.806826Z","shell.execute_reply":"2024-09-09T09:45:57.013033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for f in item_feats + ['i_full_sale','p_full_sale']:\n#     del inter[f]","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:45:57.015801Z","iopub.execute_input":"2024-09-09T09:45:57.016247Z","iopub.status.idle":"2024-09-09T09:45:57.028288Z","shell.execute_reply.started":"2024-09-09T09:45:57.016195Z","shell.execute_reply":"2024-09-09T09:45:57.026903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter['i_pop'] = popularity(base_inter, 'article_id', week_num=WEEK_NUM)\n# inter['p_pop'] = popularity(base_inter, 'product_code', week_num=WEEK_NUM)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:45:57.029625Z","iopub.execute_input":"2024-09-09T09:45:57.03003Z","iopub.status.idle":"2024-09-09T09:47:42.72008Z","shell.execute_reply.started":"2024-09-09T09:45:57.029989Z","shell.execute_reply":"2024-09-09T09:47:42.718655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter = inter.loc[inter['week'] <= WEEK_NUM + 2]","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:47:42.724104Z","iopub.execute_input":"2024-09-09T09:47:42.724709Z","iopub.status.idle":"2024-09-09T09:47:43.86767Z","shell.execute_reply.started":"2024-09-09T09:47:42.724647Z","shell.execute_reply":"2024-09-09T09:47:43.866315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inter.to_parquet(data_dir / \"processed/processed_inter.pqt\")","metadata":{"execution":{"iopub.status.busy":"2024-09-09T09:47:43.869916Z","iopub.execute_input":"2024-09-09T09:47:43.870381Z","iopub.status.idle":"2024-09-09T09:47:49.454602Z","shell.execute_reply.started":"2024-09-09T09:47:43.870336Z","shell.execute_reply":"2024-09-09T09:47:49.452817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merge Features","metadata":{}},{"cell_type":"code","source":"# inter = pd.read_parquet(data_dir / \"processed/processed_inter.pqt\")\n# inter = inter[inter['week'] <= WEEK_NUM + 2]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DRAFTS","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}