{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 52nd Place Solution Notebook\n\nThis notebook is a cleaned version of my final submission.  \nSee [this post](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/324076/) for some details about my solution.\n\nThe notebook has minimal code in it - most of the code is imported from my [handmhelpers dataset](https://www.kaggle.com/datasets/jacob34/handmhelpers), which is synced to [this github repo](https://github.com/JacobCP/kaggle-handm-helpers) .  \nSee [this post](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/324078) for some details about my code development.  \n\n**Please note:**  \nI plan on continuing to update the github repo, as I try to recreate some of the strategies shared by winning teams.  \nSome of those changes may break the code usage for this notebook.  \nIn order to keep this notebook functional, I will no longer be updating the dataset to reflect the changes made to the repo - it will remain at commit 86c412e902a7692b24e15791322a8dfeb5a761eb","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:49:17.756325Z","iopub.execute_input":"2026-09-11T10:49:17.756491Z","iopub.status.idle":"2026-09-11T10:49:17.877978Z","shell.execute_reply.started":"2026-09-11T10:49:17.756472Z","shell.execute_reply":"2026-09-11T10:49:17.877183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, sys, shutil, glob, importlib\nimport pandas as pd\nimport numpy as np\n\n# ========== 1. 复制 handmhelpers 到可写目录 ==========\nsrc = '/kaggle/input/handmhelpers'\ndst = '/kaggle/working/handmhelpers'\n\nif os.path.exists(dst):\n    shutil.rmtree(dst)\nshutil.copytree(src, dst)\nprint(f\"✅ 已复制到 {dst}\")\n\n# ========== 2. 打所有已知补丁 ==========\ndef apply_all_patches(folder):\n    patched = {}\n    for f in glob.glob(folder + '/*.py'):\n        name = os.path.basename(f)\n        with open(f, 'r') as fh:\n            content = fh.read()\n        orig = content\n\n        # 补丁 1: Series.applymap → Series.apply（pandas 新版已移除）\n        content = content.replace('.applymap(', '.apply(')\n\n        # 补丁 2: cudf 专有 .to_pandas() → 直接删（自己就是 pandas）\n        content = content.replace('.to_pandas()', '')\n\n        if content != orig:\n            with open(f, 'w') as fh:\n                fh.write(content)\n            patched[name] = True\n    return list(patched.keys())\n\npatched_files = apply_all_patches(dst)\nprint(f\"✅ 补丁文件: {patched_files}\")\n\n# ========== 3. 掉包 cudf → pandas ==========\nsys.modules['cudf'] = pd\n\n# 加上 cudf 常见的 from_pandas 接口\npd.from_pandas = lambda x, **kw: x\nif not hasattr(pd.DataFrame, 'from_pandas'):\n    pd.DataFrame.from_pandas = classmethod(lambda cls, x, **kw: x)\nif not hasattr(pd.Series, 'from_pandas'):\n    pd.Series.from_pandas = classmethod(lambda cls, x, **kw: x)\n\n# ========== 4. 清掉已缓存的 handmhelpers 模块 ==========\nfor key in list(sys.modules.keys()):\n    if key.startswith('handmhelpers'):\n        del sys.modules[key]\n\n# ========== 5. working 路径优先 ==========\nif '/kaggle/working' not in sys.path:\n    sys.path.insert(0, '/kaggle/working')\n\n# ========== 6. 从 working 导入 ==========\nfrom handmhelpers import io as h_io, sub as h_sub, cv as h_cv, fe as h_fe\nfrom handmhelpers import modeling as h_modeling, candidates as h_can, pairs as h_pairs\n\n# ========== 7. 验证 ==========\nimport cudf, handmhelpers\nprint(f\"✅ cudf 现在是: {cudf.__name__}\")\nprint(f\"✅ handmhelpers 来自: {handmhelpers.__file__}\")","metadata":{"execution":{"iopub.status.busy":"2026-09-11T10:49:22.125263Z","iopub.execute_input":"2026-09-11T10:49:22.126060Z","iopub.status.idle":"2026-09-11T10:49:28.060310Z","shell.execute_reply.started":"2026-09-11T10:49:22.126020Z","shell.execute_reply":"2026-09-11T10:49:28.059663Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and convert data","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nDATA_DIR = '/kaggle/input/h-and-m-personalized-fashion-recommendations/'\n\nc = pd.read_csv(DATA_DIR + 'customers.csv')\na = pd.read_csv(DATA_DIR + 'articles.csv')\nprint(\"✅ customers, articles 读取完毕\")\n\nchunks = []\nfor chunk in pd.read_csv(\n    DATA_DIR + 'transactions_train.csv',\n    usecols=['t_dat', 'customer_id', 'article_id', 'price', 'sales_channel_id'],\n    dtype={'price': 'float32', 'sales_channel_id': 'int8'},\n    chunksize=3_000_000\n):\n    chunk = chunk[chunk['t_dat'] >= '2020-06-01']\n    if len(chunk) > 0:\n        chunks.append(chunk)\n\nt = pd.concat(chunks, ignore_index=True)\ndel chunks\ngc.collect()\nt['t_dat'] = pd.to_datetime(t['t_dat'])\n\n# ========== 关键修改：用绝对基准 2018-09-20 计算 week_number ==========\nBASE_DATE = pd.Timestamp('2018-09-20')\nt[\"week_number\"] = ((t[\"t_dat\"] - BASE_DATE).dt.days // 7).astype('int16')\n\n# t_dat 转成相对天数（从最早一天开始算天数）\nmin_date = t[\"t_dat\"].min()\nt[\"t_dat\"] = ((t[\"t_dat\"] - min_date).dt.days).astype('int16')\n\nprint(f\"✅ 保留 {len(t):,} 行\")\nprint(f\"✅ week_number 范围: {t['week_number'].min()} ~ {t['week_number'].max()}\")\nprint(f\"✅ t_dat 范围: {t['t_dat'].min()} ~ {t['t_dat'].max()}\")\n\n# ========== 其余替换成原博主的 reduce_customer_id_memory ==========\nindex_to_id_dict_path = h_fe.reduce_customer_id_memory(c, [t])\nprint(\"✅✅✅ 数据加载完成\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:49:34.561416Z","iopub.execute_input":"2026-09-11T10:49:34.562494Z","iopub.status.idle":"2026-09-11T10:50:26.421363Z","shell.execute_reply.started":"2026-09-11T10:49:34.562464Z","shell.execute_reply":"2026-09-11T10:50:26.420650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\npairs_per_item = 5\n\nweek_number_pairs = {}\nfor week_number in [96, 97, 98, 99, 100, 101, 102, 103, 104]:\n    print(f\"Creating pairs for week number {week_number}\")\n    week_number_pairs[week_number] = h_pairs.create_pairs(\n        t, week_number, pairs_per_item, verbose=False\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:50:31.739589Z","iopub.execute_input":"2026-09-11T10:50:31.740037Z","iopub.status.idle":"2026-09-11T10:51:07.428095Z","shell.execute_reply.started":"2026-09-11T10:50:31.740003Z","shell.execute_reply":"2026-09-11T10:51:07.427399Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Main retrieval/features function!","metadata":{}},{"cell_type":"code","source":"import glob\n\ndef create_candidates_with_features_df(t, c, a, customer_batch=None, **kwargs):\n    # splitting cv\n    features_df, label_df = h_cv.feature_label_split(\n        t, kwargs[\"label_week\"], kwargs[\"feature_periods\"]\n    )\n    \n    # converting relative day_number\n    features_df[\"t_dat\"] = h_fe.how_many_ago(features_df[\"t_dat\"])\n    features_df[\"week_number\"] = h_fe.how_many_ago(features_df[\"week_number\"])\n    \n    # pull out the cv week\n    article_pairs_df = week_number_pairs[kwargs[\"label_week\"]-1]\n    \n    # check if we can limit customers\n    if len(label_df) > 0:\n        customers = label_df[\"customer_id\"].unique()\n    elif customer_batch is not None:\n        customers = customer_batch\n    else:\n        customers = None\n    \n    ############################################\n    # creating candidates (and adding features)\n    ###########################################\n    \n    features_db = shelve.open(\"features_db\")\n    \n    # creating candidate (and saving features created)\n    recent_customer_cand, features_db[\"customer_article\"] = (\n        h_can.create_recent_customer_candidates(\n            features_df,\n            kwargs[\"ca_num_weeks\"],\n            customers=customers,\n        )\n    )\n    \n    (cust_last_week_cand,\n     cust_last_week_pair_cand,\n     features_db[\"clw\"],\n     features_db[\"clw_pairs\"]) = h_can.create_last_customer_weeks_and_pairs(\n        features_df,\n        article_pairs_df,\n        kwargs[\"clw_num_weeks\"],\n        kwargs[\"clw_num_pair_weeks\"],\n        customers=customers,\n    )\n    \n    _, features_db[\"popular_articles\"] = h_can.create_popular_article_cand(\n        features_df,\n        c,\n        a,\n        kwargs[\"pa_num_weeks\"],\n        kwargs[\"hier_col\"],\n        num_candidates=kwargs[\"num_recent_candidates\"],\n        num_articles=kwargs[\"num_recent_articles\"],\n        customers=customers,\n    )\n    age_bucket_can, _, _ = h_can.create_age_bucket_candidates(\n        features_df,\n        c,\n        kwargs[\"num_age_buckets\"],\n        articles=kwargs[\"num_recent_articles\"],\n        customers=customers,\n    )\n    \n    cand = [recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can]\n    cand = cudf.concat(cand).drop_duplicates()\n    cand = cand.sort_values([\"customer_id\", \"article_id\"]).reset_index(drop=True)\n    \n    del recent_customer_cand, cust_last_week_cand, cust_last_week_pair_cand, age_bucket_can\n    \n    cand = h_can.filter_candidates(cand, t, **kwargs)\n    \n    # creating other features\n    h_fe.create_cust_hier_features(features_df, a, kwargs[\"hier_cols\"], features_db)\n    h_fe.create_price_features(features_df, features_db)\n    h_fe.create_cust_features(c, features_db)\n    h_fe.create_article_cust_features(features_df, c, features_db)\n    h_fe.create_lag_features(features_df, a, kwargs[\"lag_days\"], features_db)\n    h_fe.create_rebuy_features(features_df, features_db)\n    h_fe.create_cust_t_features(features_df, a, features_db)\n    h_fe.create_art_t_features(features_df, features_db)\n    \n    del features_df\n\n    # another filter at the end\n    if customers is not None:\n        cand = cand[cand[\"customer_id\"].isin(customers)]\n    \n    # report on recall/precision of candidates\n    if kwargs[\"cv\"]:\n        ground_truth_candidates = label_df[[\"customer_id\", \"article_id\"]].drop_duplicates()\n        h_cv.report_candidates(cand, ground_truth_candidates)\n        del ground_truth_candidates        \n    \n    # adding features to candidates\n    cand_with_f_df = h_can.add_features_to_candidates(\n        cand, features_db, c, a\n    )\n    \n    # manually adding article features\n    for article_col in kwargs[\"article_columns\"]:\n        art_col_map = a.set_index(\"article_id\")[article_col]\n        cand_with_f_df[article_col] = cand_with_f_df[\"article_id\"].map(art_col_map)\n    \n    # limiting features\n     # limiting features\n    if kwargs[\"selected_features\"] is not None:\n        cand_with_f_df = cand_with_f_df[\n            [\"customer_id\", \"article_id\"] + kwargs[\"selected_features\"]\n        ]\n    \n    # ⭐ 关键补丁：把 object 列转成 category，让 LightGBM 接受\n    for col in cand_with_f_df.columns:\n        if cand_with_f_df[col].dtype == 'object':\n            cand_with_f_df[col] = cand_with_f_df[col].astype('category')\n    \n    features_db.close()\n    assert len(cand) == len(cand_with_f_df), \"seem to have duplicates in the feature dfs\"\n    del cand\n    \n    return cand_with_f_df, label_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:51:12.668562Z","iopub.execute_input":"2026-09-11T10:51:12.668883Z","iopub.status.idle":"2026-09-11T10:51:12.682901Z","shell.execute_reply.started":"2026-09-11T10:51:12.668858Z","shell.execute_reply":"2026-09-11T10:51:12.682197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_model_score(ids_df, preds, truth_df):\n    predictions = h_modeling.create_predictions(ids_df, preds)\n    true_labels = h_cv.ground_truth(truth_df).set_index(\"customer_id\")[\"prediction\"]\n    score = round(h_cv.comp_average_precision(true_labels, predictions),5)\n    \n    return score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:51:24.121139Z","iopub.execute_input":"2026-09-11T10:51:24.121789Z","iopub.status.idle":"2026-09-11T10:51:24.126112Z","shell.execute_reply.started":"2026-09-11T10:51:24.121759Z","shell.execute_reply":"2026-09-11T10:51:24.125062Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Parameters - one place for all!","metadata":{}},{"cell_type":"code","source":"cv_params = {\n    \"cv\": True,\n    \"feature_periods\": 105,\n    \"label_week\": 104,\n    \"index_to_id_dict_path\": index_to_id_dict_path,\n    \"pairs_file_version\": \"_v3_5_ex\",\n    \"num_recent_candidates\": 20,      # 从 36 降低到 20\n    \"num_recent_articles\": 10,        # 从 12 降低到 10\n    \"hier_col\": \"department_no\",\n    \"ca_num_weeks\": 3,\n    \"clw_num_weeks\": 8,               # 从 12 降低到 8\n    \"clw_num_pair_weeks\": 2,\n    \"pa_num_weeks\": 1,\n    \"num_age_buckets\": 2,             # 从 4 降低到 2\n    \"filter_recent_art_weeks\": 1,\n    \"filter_num_articles\": None,\n    \"lag_days\": [1, 3, 14, 30],\n    \"article_columns\": [\"index_code\"],\n    \"hier_cols\": [\"department_no\", \"section_no\", \"index_group_no\", \"index_code\",\n                  \"product_type_no\", \"product_group_name\"],\n    \"selected_features\": None,\n    \"lgbm_params\": {\"n_estimators\": 200, \"num_leaves\": 20},\n    \"log_evaluation\": 10,\n    \"early_stopping\": 20,\n    \"eval_at\": 12,\n    \"save_model\": True,\n    \"num_concats\": 5,\n}\n\n# sub_params 的修改类似，请参考原代码进行调整\nsub_params = {\n    \"cv\": False,\n    \"feature_periods\": 105,\n    \"label_week\": 105,\n    \"index_to_id_dict_path\": index_to_id_dict_path,\n    \"pairs_file_version\": \"_v3_5_ex\",\n    \"num_recent_candidates\": 60,\n    \"num_recent_articles\": 12,\n    \"hier_col\": \"department_no\",\n    \"ca_num_weeks\": 3,\n    \"clw_num_weeks\": 12,\n    \"clw_num_pair_weeks\": 2,\n    \"pa_num_weeks\": 1,\n    \"num_age_buckets\": 4,\n    \"filter_recent_art_weeks\": 1,\n    \"filter_num_articles\": None,\n    \"lag_days\": [1, 3, 14, 30],\n    \"article_columns\": [\"index_code\"],\n    \"hier_cols\": [\"department_no\", \"section_no\", \"index_group_no\", \"index_code\",\n                  \"product_type_no\", \"product_group_name\"],\n    \"selected_features\": None,\n    \"lgbm_params\": {\n        \"n_estimators\": 150,\n        \"num_leaves\": 20,    \n    },\n    \"log_evaluation\": 10,\n    \"eval_at\": 12,\n    \"prediction_models\": [\"model_104\", \"model_105\"],\n    \"save_model\": True,\n    \"num_concats\": 5,\n}\n\ncand_features_func = create_candidates_with_features_df\nscoring_func = calculate_model_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:51:27.378351Z","iopub.execute_input":"2026-09-11T10:51:27.378799Z","iopub.status.idle":"2026-09-11T10:51:27.385550Z","shell.execute_reply.started":"2026-09-11T10:51:27.378773Z","shell.execute_reply":"2026-09-11T10:51:27.384924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cand_features_func = create_candidates_with_features_df\nscoring_func = calculate_model_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:51:30.974612Z","iopub.execute_input":"2026-09-11T10:51:30.975291Z","iopub.status.idle":"2026-09-11T10:51:30.978912Z","shell.execute_reply.started":"2026-09-11T10:51:30.975261Z","shell.execute_reply":"2026-09-11T10:51:30.978252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shelve","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:51:34.383914Z","iopub.execute_input":"2026-09-11T10:51:34.384217Z","iopub.status.idle":"2026-09-11T10:51:34.389754Z","shell.execute_reply.started":"2026-09-11T10:51:34.384192Z","shell.execute_reply":"2026-09-11T10:51:34.389150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob, sys\n\ndst = '/kaggle/working/handmhelpers'\n\n# 补丁：.astype(\"int16\").fillna(-1)  →  .fillna(-1).astype(\"int16\")\npatched = []\nfor f in glob.glob(dst + '/*.py'):\n    with open(f, 'r') as fh:\n        content = fh.read()\n    orig = content\n\n    # 通用写法：把 .astype(\"xxx\").fillna(YYY) 颠倒过来\n    # 用正则匹配常见 dtype\n    import re\n    content = re.sub(\n        r'\\.astype\\([\"\\']int(8|16|32|64)[\"\\']\\)\\.fillna\\(([^)]+)\\)',\n        r'.fillna(\\2).astype(\"int\\1\")',\n        content\n    )\n\n    if content != orig:\n        with open(f, 'w') as fh:\n            fh.write(content)\n        patched.append(f.split('/')[-1])\n\nprint(f\"✅ 修补文件: {patched}\")\n\n# 彻底清掉已缓存的 handmhelpers 模块\nfor key in list(sys.modules.keys()):\n    if key.startswith('handmhelpers'):\n        del sys.modules[key]\n\n# 重新导入\nfrom handmhelpers import io as h_io, sub as h_sub, cv as h_cv, fe as h_fe\nfrom handmhelpers import modeling as h_modeling, candidates as h_can, pairs as h_pairs\nprint(\"✅ 重新导入完成\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:51:36.567164Z","iopub.execute_input":"2026-09-11T10:51:36.567825Z","iopub.status.idle":"2026-09-11T10:51:36.584013Z","shell.execute_reply.started":"2026-09-11T10:51:36.567794Z","shell.execute_reply":"2026-09-11T10:51:36.583040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, pickle, shelve\n\nclass FileShelve:\n    def __init__(self, base):\n        self.base = base\n        self.dir = os.path.dirname(base) or \".\"\n        self.prefix = os.path.basename(base) + \"_\"\n\n    def _path(self, key):\n        return os.path.join(self.dir, f\"{self.prefix}{key}.pkl\")\n\n    def __setitem__(self, key, value):\n        with open(self._path(key), \"wb\") as f:\n            pickle.dump(value, f, protocol=4)\n\n    def __getitem__(self, key):\n        with open(self._path(key), \"rb\") as f:\n            return pickle.load(f)\n\n    def __contains__(self, key):\n        return os.path.exists(self._path(key))\n\n    def __delitem__(self, key):\n        if os.path.exists(self._path(key)):\n            os.remove(self._path(key))\n\n    def __iter__(self):\n        for fname in os.listdir(self.dir):\n            if fname.startswith(self.prefix) and fname.endswith(\".pkl\"):\n                yield fname[len(self.prefix):-4]\n\n    def keys(self):\n        return list(self.__iter__())\n\n    def __len__(self):\n        return len(self.keys())\n\n    def close(self):\n        pass\n\nshelve.open = lambda path, **kw: FileShelve(path)\nprint(\"✅ FileShelve 已更新（支持迭代）\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:51:43.123037Z","iopub.execute_input":"2026-09-11T10:51:43.123716Z","iopub.status.idle":"2026-09-11T10:51:43.132421Z","shell.execute_reply.started":"2026-09-11T10:51:43.123676Z","shell.execute_reply":"2026-09-11T10:51:43.131695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ncv_weeks = [104]\nresults = h_modeling.run_all_cvs(\n    t, c, a, cand_features_func, scoring_func, \n    cv_weeks=cv_weeks, **cv_params\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:51:47.178826Z","iopub.execute_input":"2026-09-11T10:51:47.179505Z","iopub.status.idle":"2026-09-11T10:56:06.842858Z","shell.execute_reply.started":"2026-09-11T10:51:47.179473Z","shell.execute_reply":"2026-09-11T10:56:06.842019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ngc.collect()\nh_modeling.full_sub_train_run(t, c, a, cand_features_func, scoring_func, **sub_params)\npredictions = h_modeling.full_sub_predict_run(\n    t, c, a, cand_features_func, **sub_params\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T10:56:13.229052Z","iopub.execute_input":"2026-09-11T10:56:13.229435Z","iopub.status.idle":"2026-09-11T11:07:59.811377Z","shell.execute_reply.started":"2026-09-11T10:56:13.229407Z","shell.execute_reply":"2026-09-11T11:07:59.810507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = h_sub.create_sub(c[\"customer_id\"], predictions, index_to_id_dict_path)\nsub.to_csv('dev_submission.csv', index=False)\n\ndisplay(sub.head())\nprint(sub.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T11:08:25.676580Z","iopub.execute_input":"2026-09-11T11:08:25.677068Z","iopub.status.idle":"2026-09-11T11:08:52.965187Z","shell.execute_reply.started":"2026-09-11T11:08:25.677039Z","shell.execute_reply":"2026-09-11T11:08:52.964518Z"}},"outputs":[],"execution_count":null}]}