{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":3618498,"sourceType":"datasetVersion","datasetId":1931827},{"sourceId":94922230,"sourceType":"kernelVersion"}],"dockerImageVersionId":30153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 52nd Place Solution Notebook\n\nThis notebook is a cleaned version of my final submission.  \nSee [this post](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/324076/) for some details about my solution.\n\nThe notebook has minimal code in it - most of the code is imported from my [handmhelpers dataset](https://www.kaggle.com/datasets/jacob34/handmhelpers), which is synced to [this github repo](https://github.com/JacobCP/kaggle-handm-helpers) .  \nSee [this post](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/324078) for some details about my code development.  \n\n**Please note:**  \nI plan on continuing to update the github repo, as I try to recreate some of the strategies shared by winning teams.  \nSome of those changes may break the code usage for this notebook.  \nIn order to keep this notebook functional, I will no longer be updating the dataset to reflect the changes made to the repo - it will remain at commit 86c412e902a7692b24e15791322a8dfeb5a761eb","metadata":{}},{"cell_type":"markdown","source":"I developed this notebook basing on [this notebook](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/324076/).","metadata":{}},{"cell_type":"code","source":"%%time\nimport os\nimport sys\nimport copy\nfrom datetime import datetime\nimport gc\nimport pickle as pkl\nimport shelve\n\nimport pandas as pd\nimport numpy as np\nimport cudf\n    \nsys.path.append(\"../input/\")\n#https://github.com/JacobCP/kaggle-handm-helpers\nfrom handmhelpers import io as h_io, sub as h_sub, cv as h_cv, fe as h_fe\nfrom handmhelpers import modeling as h_modeling, candidates as h_can, pairs as h_pairs","metadata":{"execution":{"iopub.status.busy":"2025-01-30T14:05:10.620039Z","iopub.execute_input":"2025-01-30T14:05:10.620951Z","iopub.status.idle":"2025-01-30T14:05:15.939353Z","shell.execute_reply.started":"2025-01-30T14:05:10.620897Z","shell.execute_reply":"2025-01-30T14:05:15.938556Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and convert data","metadata":{}},{"cell_type":"code","source":"%%time\n\nc, t, a = h_io.load_data(files=['customers.csv', 'transactions_train.csv', 'articles.csv'])        \n\nindex_to_id_dict_path = h_fe.reduce_customer_id_memory(c, [t])\nt[\"week_number\"] = h_fe.day_week_numbers(t[\"t_dat\"])\nt[\"t_dat\"] = h_fe.day_numbers(t[\"t_dat\"])","metadata":{"execution":{"iopub.status.busy":"2025-01-30T14:10:56.964540Z","iopub.execute_input":"2025-01-30T14:10:56.965324Z","iopub.status.idle":"2025-01-30T14:11:49.621504Z","shell.execute_reply.started":"2025-01-30T14:10:56.965289Z","shell.execute_reply":"2025-01-30T14:11:49.620804Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir(os.getcwd())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-30T14:12:15.057752Z","iopub.execute_input":"2025-01-30T14:12:15.058362Z","iopub.status.idle":"2025-01-30T14:12:15.063674Z","shell.execute_reply.started":"2025-01-30T14:12:15.058331Z","shell.execute_reply":"2025-01-30T14:12:15.063018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for data in [c,t,a]:\n    print(\"\")\n    print(data.head(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-30T14:14:29.045377Z","iopub.execute_input":"2025-01-30T14:14:29.045974Z","iopub.status.idle":"2025-01-30T14:14:29.638571Z","shell.execute_reply.started":"2025-01-30T14:14:29.045944Z","shell.execute_reply":"2025-01-30T14:14:29.637658Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Get item pairs","metadata":{}},{"cell_type":"code","source":"%%time\n#统计每个商品和其他商品被同时购买的数量和占比（占购买该商品总人数的比例）\n#pair商品的数量\npairs_per_item = 5\n\nweek_number_pairs = {}\nfor week_number in [96, 97, 98, 99, 100, 101, 102, 103, 104]:\n    print(f\"Creating pairs for week number {week_number}\")\n    week_number_pairs[week_number] = h_pairs.create_pairs(\n        t, week_number, pairs_per_item, verbose=False\n    )","metadata":{"execution":{"iopub.status.busy":"2025-01-30T14:16:36.155472Z","iopub.execute_input":"2025-01-30T14:16:36.156201Z","iopub.status.idle":"2025-01-30T14:17:41.471324Z","shell.execute_reply.started":"2025-01-30T14:16:36.156169Z","shell.execute_reply":"2025-01-30T14:17:41.470542Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"week_number_pairs[97]['percent_customers'].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T14:29:24.366009Z","iopub.execute_input":"2025-01-29T14:29:24.366379Z","iopub.status.idle":"2025-01-29T14:29:24.378934Z","shell.execute_reply.started":"2025-01-29T14:29:24.366342Z","shell.execute_reply":"2025-01-29T14:29:24.378304Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Main retrieval/features function!","metadata":{}},{"cell_type":"markdown","source":"```python\ncand = [recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can]\n```\n召回通道的设置：\n* 最近购买物品召回：将用户最近三周购买的物品按照购买次数降序排列，\n* 购买时间召回：将距离最近一次购买时间（不计购买什么物品）小于十二周的商品召回。\n* 共同物品召回（u2u召回）：统计物品a和物品b被同一用户购买的次数（过去一周的购买次数），过滤掉购买次数小于2的商品，召回距离最近一次购买小于2周的商品拿出来，取他们的共同物品做召回。\n* 年龄分桶热度召回：将用户按照年龄分桶，然后分年龄桶统计过去一周每个商品的购买次数，每个年龄桶召回12个高热商品。\n\n将四个召回通道的结果直接合并去重组成最后的召回结果，然后再根据购买热度（最近一周商品的购买次数）过滤以达到指定数量的被召回数目（*最简单的情况下是删除掉所有没有被购买的商品，也可以根据购买次数降序排列然后截断。*）。\n\n召回效果评估：\n取最新一周的数据用户购买过的商品作为ground truth,计算召回率（分母是所有最近一周被购买的商品）和准确率（分母是召回通道的所有商品）。召回通道的平均召回率在7.4%，精度在0.8%。","metadata":{}},{"cell_type":"code","source":"def create_candidates_with_features_df(t, c, a, customer_batch=None, **kwargs):\n    #split transaction_df into train/validation\n    # feature_label_split：label_week的交易数据为validation，label_week之前的数据为train\n    # 也可以指定feature_periods\n    # 貌似label会有多个？有点不太确定\n    features_df, label_df = h_cv.feature_label_split(\n        t, kwargs[\"label_week\"], kwargs[\"feature_periods\"]\n    )\n    \n    # converting relative day_number（减去列值的最大值）\n    features_df[\"t_dat\"] = h_fe.how_many_ago(features_df[\"t_dat\"])\n    features_df[\"week_number\"] = h_fe.how_many_ago(features_df[\"week_number\"])\n    \n    # pull out the cv week\n    article_pairs_df = week_number_pairs[kwargs[\"label_week\"]-1]\n    \n    # check if we can limit customers\n    if len(label_df) > 0:\n        customers = label_df[\"customer_id\"].unique()\n    elif customer_batch is not None:\n        customers = customer_batch\n    else:\n        customers = None\n    \n    ############################################\n    # creating candidates (and adding features)\n    ###########################################\n    \n    features_db = shelve.open(\"features_db\") \n    \n    # creating candidate (and saving features created) \n    # 筛选出，同时记录购买次数\n    # recent_customer_cand: 在ca_num_weeks后每个用户有购买过的商品\n    # features_db[\"customer_article\"]:保存元祖对象，([\"customer_id\", \"article_id\"], recent_customer_df),recent_customer_df：存储了用户购买过的商品id以及购买的数量\n    recent_customer_cand, features_db[\"customer_article\"] = (\n        h_can.create_recent_customer_candidates(\n            features_df,\n            kwargs[\"ca_num_weeks\"],\n            customers=customers,#参与统计的用户名单\n        )\n    )\n    #过滤掉一批太久没有购买的商品以及它的配对商品（商品对和商品采用不同的过滤周期，对应clw_num_weeks和clw_num_pair_weeks参数\n    (cust_last_week_cand,\n     cust_last_week_pair_cand,\n     features_db[\"clw\"],\n     features_db[\"clw_pairs\"]) = h_can.create_last_customer_weeks_and_pairs(\n        features_df,\n        article_pairs_df,\n        kwargs[\"clw_num_weeks\"],\n        kwargs[\"clw_num_pair_weeks\"],\n        customers=customers,\n    )\n    # 热门商品\n    _, features_db[\"popular_articles\"] = h_can.create_popular_article_cand(\n        features_df,\n        c,\n        a,\n        kwargs[\"pa_num_weeks\"],# \n        kwargs[\"hier_col\"],\n        num_candidates=kwargs[\"num_recent_candidates\"],\n        num_articles=kwargs[\"num_recent_articles\"],# 返回热门物品的数量\n        customers=customers,\n    )\n    # 将用户的年龄分桶，然后统计每个候选商品在不同年龄段的下单量（article_bucket_count）\n    age_bucket_can, _, _ = h_can.create_age_bucket_candidates(\n        features_df,\n        c,\n        kwargs[\"num_age_buckets\"],\n        articles=kwargs[\"num_recent_articles\"],\n        customers=customers,\n    )\n    #几个召回列表都是customer_id,article_id的形式    \n    cand = [recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can]\n    cand = cudf.concat(cand).drop_duplicates()#直接合并成一个召回列表\n    cand = cand.sort_values([\"customer_id\", \"article_id\"]).reset_index(drop=True)\n    \n    del recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can\n    \n    cand = h_can.filter_candidates(cand, t, **kwargs)\n    \n    # creating other features\n    h_fe.create_cust_hier_features(features_df, a, kwargs[\"hier_cols\"], features_db)\n    h_fe.create_price_features(features_df, features_db)\n    h_fe.create_cust_features(c, features_db)\n    h_fe.create_article_cust_features(features_df, c, features_db)\n    h_fe.create_lag_features(features_df, a, kwargs[\"lag_days\"], features_db)\n    h_fe.create_rebuy_features(features_df, features_db)\n    h_fe.create_cust_t_features(features_df, a, features_db)\n    h_fe.create_art_t_features(features_df, features_db)\n    \n    del features_df\n\n    # another filter at the end, for the ones that didn't get filtered earlier\n    if customers is not None:\n        cand = cand[cand[\"customer_id\"].isin(customers)]\n    \n    # report on recall/precision of candidates\n    if kwargs[\"cv\"]:\n        ground_truth_candidates = label_df[[\"customer_id\", \"article_id\"]].drop_duplicates()\n        h_cv.report_candidates(cand, ground_truth_candidates)\n        del ground_truth_candidates        \n    \n    # adding features to candidates\n    cand_with_f_df = h_can.add_features_to_candidates(\n        cand, features_db, c, a\n    )\n    \n    # manually adding article features (couldn't use shelve for some reason)\n    for article_col in kwargs[\"article_columns\"]:\n        art_col_map = a.set_index(\"article_id\")[article_col]\n        cand_with_f_df[article_col] = cand_with_f_df[\"article_id\"].map(art_col_map)\n    \n    # limiting features\n    if kwargs[\"selected_features\"] is not None:\n        cand_with_f_df = cand_with_f_df[\n            [\"customer_id\", \"article_id\"] + kwargs[\"selected_features\"]\n        ]\n        \n    features_db.close()\n    # os.remove(\"features_db.bak\"), os.remove(\"features_db.dir\"), os.remove(\"features_db.dat\")\n    \n    assert len(cand) == len(cand_with_f_df), \"seem to have duplicates in the feature dfs\"\n    del cand\n    \n    return cand_with_f_df, label_df","metadata":{"execution":{"iopub.status.busy":"2025-01-30T14:17:56.155380Z","iopub.execute_input":"2025-01-30T14:17:56.155611Z","iopub.status.idle":"2025-01-30T14:17:56.169619Z","shell.execute_reply.started":"2025-01-30T14:17:56.155588Z","shell.execute_reply":"2025-01-30T14:17:56.168818Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_model_score(ids_df, preds, truth_df):\n    predictions = h_modeling.create_predictions(ids_df, preds)\n    true_labels = h_cv.ground_truth(truth_df).set_index(\"customer_id\")[\"prediction\"]\n    score = round(h_cv.comp_average_precision(true_labels, predictions),5)\n    \n    return score\n    ","metadata":{"execution":{"iopub.status.busy":"2025-01-30T14:19:44.721596Z","iopub.execute_input":"2025-01-30T14:19:44.722203Z","iopub.status.idle":"2025-01-30T14:19:44.726742Z","shell.execute_reply.started":"2025-01-30T14:19:44.722165Z","shell.execute_reply":"2025-01-30T14:19:44.725926Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Parameters - one place for all!","metadata":{}},{"cell_type":"code","source":"#交叉验证是以一周为验证集，该周前边的数据为训练集\ncv_params = {\n    \"cv\": True,\n    \"feature_periods\": 105,#训练集的周数\n    \"label_week\": 104,# 验证集的周\n    \"index_to_id_dict_path\": index_to_id_dict_path,\n    \"pairs_file_version\": \"_v3_5_ex\",\n    \"num_recent_candidates\": 36,\n    \"num_recent_articles\": 12,\n    \"hier_col\": \"department_no\",\n    \"ca_num_weeks\": 3,#统计用户购买商品的时间窗大小\n    \"clw_num_weeks\": 12,\n    \"clw_num_pair_weeks\": 2,\n    \"pa_num_weeks\": 1,\n    \"num_age_buckets\": 4,\n    \"filter_recent_art_weeks\": 1,\n    \"filter_num_articles\": None,\n    \"lag_days\": [1, 3, 14, 30],\n    \"article_columns\": [\"index_code\"],\n    \"hier_cols\": [\n        \"department_no\", \"section_no\", \"index_group_no\", \"index_code\",\n        \"product_type_no\", \"product_group_name\"\n    ],\n    \"selected_features\": None,\n    \"lgbm_params\": {\"n_estimators\": 200, \"num_leaves\": 20},\n    \"log_evaluation\": 10,\n    \"early_stopping\": 20,\n    \"eval_at\": 12,\n    \"save_model\": True,\n    \"num_concats\": 5,\n}\nsub_params = {\n    \"cv\": False,\n    \"feature_periods\": 105,\n    \"label_week\": 105,\n    \"index_to_id_dict_path\": index_to_id_dict_path,\n    \"pairs_file_version\": \"_v3_5_ex\",\n    \"num_recent_candidates\": 60,\n    \"num_recent_articles\": 12,\n    \"hier_col\": \"department_no\",\n    \"ca_num_weeks\": 3,\n    \"clw_num_weeks\": 12,\n    \"clw_num_pair_weeks\": 2,\n    \"pa_num_weeks\": 1,\n    \"num_age_buckets\": 4,\n    \"filter_recent_art_weeks\": 1,\n    \"filter_num_articles\": None,\n    \"lag_days\": [1, 3, 14, 30],\n    \"article_columns\": [\"index_code\"],\n    \"hier_cols\": [\n        \"department_no\", \"section_no\", \"index_group_no\", \"index_code\",\n        \"product_type_no\", \"product_group_name\"\n    ],\n    \"selected_features\": None,\n    \"lgbm_params\": {\n        \"n_estimators\": 150,\n        \"num_leaves\": 20,    \n    },\n    \"log_evaluation\": 10,\n    \"eval_at\": 12,\n    \"prediction_models\": [\"model_104\", \"model_105\"],\n    \"save_model\": True,\n    \"num_concats\": 5,\n}","metadata":{"execution":{"iopub.status.busy":"2025-01-30T14:19:48.938597Z","iopub.execute_input":"2025-01-30T14:19:48.939282Z","iopub.status.idle":"2025-01-30T14:19:48.947042Z","shell.execute_reply.started":"2025-01-30T14:19:48.939245Z","shell.execute_reply":"2025-01-30T14:19:48.946343Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cand_features_func = create_candidates_with_features_df\nscoring_func = calculate_model_score","metadata":{"execution":{"iopub.status.busy":"2025-01-30T14:20:00.952871Z","iopub.execute_input":"2025-01-30T14:20:00.953373Z","iopub.status.idle":"2025-01-30T14:20:00.956819Z","shell.execute_reply.started":"2025-01-30T14:20:00.953340Z","shell.execute_reply":"2025-01-30T14:20:00.956108Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ncv_weeks = [104]\nresults = h_modeling.run_all_cvs(\n    t, c, a, cand_features_func, scoring_func, \n    cv_weeks=cv_weeks, **cv_params\n)","metadata":{"execution":{"iopub.status.busy":"2025-01-30T14:20:01.635464Z","iopub.execute_input":"2025-01-30T14:20:01.635990Z","iopub.status.idle":"2025-01-30T14:22:26.877435Z","shell.execute_reply.started":"2025-01-30T14:20:01.635957Z","shell.execute_reply":"2025-01-30T14:22:26.876690Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir(os.getcwd())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-30T14:25:45.030575Z","iopub.execute_input":"2025-01-30T14:25:45.030845Z","iopub.status.idle":"2025-01-30T14:25:45.036836Z","shell.execute_reply.started":"2025-01-30T14:25:45.030823Z","shell.execute_reply":"2025-01-30T14:25:45.036120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ngc.collect()\nh_modeling.full_sub_train_run(t, c, a, cand_features_func, scoring_func, **sub_params)\npredictions = h_modeling.full_sub_predict_run(\n    t, c, a, cand_features_func, **sub_params\n)","metadata":{"execution":{"iopub.status.busy":"2025-01-29T14:32:12.331736Z","iopub.execute_input":"2025-01-29T14:32:12.332005Z","iopub.status.idle":"2025-01-29T14:40:07.693071Z","shell.execute_reply.started":"2025-01-29T14:32:12.331968Z","shell.execute_reply":"2025-01-29T14:40:07.692380Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = h_sub.create_sub(c[\"customer_id\"], predictions, index_to_id_dict_path)\nsub.to_csv('dev_submission.csv', index=False)\n\ndisplay(sub.head())\nprint(sub.shape)","metadata":{"execution":{"iopub.status.busy":"2025-01-29T14:40:07.694170Z","iopub.execute_input":"2025-01-29T14:40:07.694410Z","iopub.status.idle":"2025-01-29T14:41:11.614762Z","shell.execute_reply.started":"2025-01-29T14:40:07.694383Z","shell.execute_reply":"2025-01-29T14:41:11.614031Z"},"trusted":true},"outputs":[],"execution_count":null}]}