{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14112508,"sourceType":"datasetVersion","datasetId":8989713},{"sourceId":14230927,"sourceType":"datasetVersion","datasetId":8622232},{"sourceId":14241442,"sourceType":"datasetVersion","datasetId":9085906},{"sourceId":14399697,"sourceType":"datasetVersion","datasetId":9048408},{"sourceId":14400522,"sourceType":"datasetVersion","datasetId":9051643},{"sourceId":14403608,"sourceType":"datasetVersion","datasetId":9189582,"isSourceIdPinned":true},{"sourceId":14444034,"sourceType":"datasetVersion","datasetId":9226352},{"sourceId":14444057,"sourceType":"datasetVersion","datasetId":9218926,"isSourceIdPinned":false},{"sourceId":14408814,"sourceType":"datasetVersion","datasetId":9189780,"isSourceIdPinned":true}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import ndcg_score\nimport gc\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.778241Z","iopub.execute_input":"2026-01-09T17:15:54.778889Z","iopub.status.idle":"2026-01-09T17:15:54.786288Z","shell.execute_reply.started":"2026-01-09T17:15:54.778850Z","shell.execute_reply":"2026-01-09T17:15:54.784992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"WEEK_VALID= 1\nEND_WEEK = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.788574Z","iopub.execute_input":"2026-01-09T17:15:54.788934Z","iopub.status.idle":"2026-01-09T17:15:54.806290Z","shell.execute_reply.started":"2026-01-09T17:15:54.788902Z","shell.execute_reply":"2026-01-09T17:15:54.805132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"NUM = 12","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.807657Z","iopub.execute_input":"2026-01-09T17:15:54.808081Z","iopub.status.idle":"2026-01-09T17:15:54.828982Z","shell.execute_reply.started":"2026-01-09T17:15:54.808040Z","shell.execute_reply":"2026-01-09T17:15:54.827794Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Valid","metadata":{}},{"cell_type":"markdown","source":"v8 pri 0.0247 pub 0.025 \\\nv9 pri 0.02497 pub 0.02531 \\\nv7 pri 0.02043 pub 0.02043 \\\nV6 PRI 0.01988 PUB 0.01969","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ranker_lightgbm_valid= pd.read_parquet('/kaggle/input/hm-ranker-lightgbm-test/valid_ranker_lgbm.pqt')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.831472Z","iopub.execute_input":"2026-01-09T17:15:54.831823Z","iopub.status.idle":"2026-01-09T17:15:54.851163Z","shell.execute_reply.started":"2026-01-09T17:15:54.831783Z","shell.execute_reply":"2026-01-09T17:15:54.849950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ranker_lightgbm_valid=ranker_lightgbm_valid.sort_values(by=['customer_id', 'prob'], ascending=False).reset_index(drop=True)\n# ranker_lightgbm_valid=ranker_lightgbm_valid.groupby('customer_id')['prediction'].apply(list).reset_index()\n# ranker_lightgbm_valid = ranker_lightgbm_valid.rename(columns={'prediction': 'lgbm_rank'})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.852358Z","iopub.execute_input":"2026-01-09T17:15:54.852629Z","iopub.status.idle":"2026-01-09T17:15:54.871353Z","shell.execute_reply.started":"2026-01-09T17:15:54.852604Z","shell.execute_reply":"2026-01-09T17:15:54.870217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# binary_lightgbm_valid= pd.read_parquet('/kaggle/input/h-and-m-binary-lightgbm-test/valid_binary_lgbm.pqt')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.872871Z","iopub.execute_input":"2026-01-09T17:15:54.873276Z","iopub.status.idle":"2026-01-09T17:15:54.894199Z","shell.execute_reply.started":"2026-01-09T17:15:54.873241Z","shell.execute_reply":"2026-01-09T17:15:54.892998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# binary_lightgbm_valid=binary_lightgbm_valid.sort_values(by=['customer_id', 'prob'], ascending=False).reset_index(drop=True)\n# binary_lightgbm_valid=binary_lightgbm_valid.groupby('customer_id')['prediction'].apply(list).reset_index()\n# binary_lightgbm_valid = binary_lightgbm_valid.rename(columns={'prediction': 'lgbm_binary'})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.895761Z","iopub.execute_input":"2026-01-09T17:15:54.896164Z","iopub.status.idle":"2026-01-09T17:15:54.922143Z","shell.execute_reply.started":"2026-01-09T17:15:54.896123Z","shell.execute_reply":"2026-01-09T17:15:54.920765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred = pd.merge(binary_lightgbm_valid, ranker_lightgbm_valid, on='customer_id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.923900Z","iopub.execute_input":"2026-01-09T17:15:54.924668Z","iopub.status.idle":"2026-01-09T17:15:54.945329Z","shell.execute_reply.started":"2026-01-09T17:15:54.924619Z","shell.execute_reply":"2026-01-09T17:15:54.944052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cust_blend(dt, W=[1, 1]):\n    # Create a list of all model predictions\n    REC = []\n\n    # Add lgbm_rank and lgbm_binary to the list of recommendations\n    REC.append(dt['lgbm_rank'])\n    REC.append(dt['lgbm_binary'])\n\n    # Create a dictionary of items recommended.\n    # Assign a weight according to the order of appearance and multiply by global weights\n    res = {}\n    for M in range(len(REC)):\n        for n, v in enumerate(REC[M]):\n            if v in res:\n                res[v] += (W[M] / (n + 1))\n            else:\n                res[v] = (W[M] / (n + 1))\n\n    # Sort dictionary by item weights\n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n\n    # Return the top 12 items only\n    return res[:NUM]\n\nprint(\"The 'cust_blend' function has been updated with weighted blending logic.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.948391Z","iopub.execute_input":"2026-01-09T17:15:54.948750Z","iopub.status.idle":"2026-01-09T17:15:54.978194Z","shell.execute_reply.started":"2026-01-09T17:15:54.948717Z","shell.execute_reply":"2026-01-09T17:15:54.977048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:54.979240Z","iopub.execute_input":"2026-01-09T17:15:54.979563Z","iopub.status.idle":"2026-01-09T17:15:54.998069Z","shell.execute_reply.started":"2026-01-09T17:15:54.979537Z","shell.execute_reply":"2026-01-09T17:15:54.997042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred['prediction'] = pred.apply(lambda row: cust_blend(row, W=[1, 1]), axis=1)\n# print(\"The 'cust_blend' function has been applied to create 'blended_recommendations' column.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:55.000292Z","iopub.execute_input":"2026-01-09T17:15:55.001133Z","iopub.status.idle":"2026-01-09T17:15:55.020431Z","shell.execute_reply.started":"2026-01-09T17:15:55.001096Z","shell.execute_reply":"2026-01-09T17:15:55.019013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from typing import Iterable\nimport numpy as np\n\ndef _ap_at_k(actual, predicted, k=10):\n    if len(predicted) > k:\n        predicted = predicted[:k]\n\n    score = 0.0\n    num_hits = 0.0\n\n    if not actual: # Handle empty actual list: no actual items, so AP is 0\n        return 0.0\n\n    for i, p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            num_hits += 1.0\n            score += num_hits / (i + 1.0)\n\n    return score / min(len(actual), k)\n\n\ndef map_at_k(actual: Iterable, predicted: Iterable, k: int = 12) -> float:\n    \"\"\"Compute mean average precision @ k.\n\n    Parameters\n    ----------\n    actual : Iterable\n        Label.\n    predicted : Iterable\n        Predictions.\n    k : int, optional\n        k, by default ``12``.\n\n    Returns\n    -------\n    float\n        MAP@k.\n    \"\"\"\n    return np.mean(\n        [_ap_at_k(a, p, k) for a, p in zip(actual, predicted) if a is not None]\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:55.021971Z","iopub.execute_input":"2026-01-09T17:15:55.022395Z","iopub.status.idle":"2026-01-09T17:15:55.041699Z","shell.execute_reply.started":"2026-01-09T17:15:55.022354Z","shell.execute_reply":"2026-01-09T17:15:55.040176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data = pd.read_parquet(f'/kaggle/input/h-and-m-fe-dataset/enriched_data/enriched_week1_candidate.pqt')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:55.043042Z","iopub.execute_input":"2026-01-09T17:15:55.043415Z","iopub.status.idle":"2026-01-09T17:15:55.066845Z","shell.execute_reply.started":"2026-01-09T17:15:55.043384Z","shell.execute_reply":"2026-01-09T17:15:55.065231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data = data.sort_values(by=['customer_id'], ascending=False).reset_index(drop=True)\n# data = data[data['target'] == 1][['customer_id','article_id','target']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:55.068212Z","iopub.execute_input":"2026-01-09T17:15:55.068685Z","iopub.status.idle":"2026-01-09T17:15:55.087681Z","shell.execute_reply.started":"2026-01-09T17:15:55.068649Z","shell.execute_reply":"2026-01-09T17:15:55.086476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# actual_processed = data[data['target'] == 1].groupby('customer_id')['article_id'].apply(list).reset_index()\n# actual_processed.rename(columns={'article_id': 'actual'}, inplace=True)\n\n# # Merge predictions with actuals based on customer_id\n# eval_df = pd.merge(actual_processed, pred, on='customer_id', how='left')\n\n# # Fill NaN predictions with empty lists for customers with no predictions\n# eval_df['prediction'] = eval_df['prediction'].apply(lambda x: [] if not isinstance(x, list) else x)\n\n# # Extract actual and predicted lists for evaluation\n# actual_list = eval_df['actual'].tolist()\n# predicted_list = eval_df['prediction'].tolist()\n\n# # Calculate MAP@12 and Recall@12\n# map_result = map_at_k(actual_list, predicted_list, k=12)\n\n# print(f\"MAP@12: {map_result}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:55.089144Z","iopub.execute_input":"2026-01-09T17:15:55.089578Z","iopub.status.idle":"2026-01-09T17:15:55.108648Z","shell.execute_reply.started":"2026-01-09T17:15:55.089547Z","shell.execute_reply":"2026-01-09T17:15:55.107628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# del data, binary_lightgbm_valid,ranker_lightgbm_valid,actual_processed,eval_df,actual_list,predicted_list\n# gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:55.111472Z","iopub.execute_input":"2026-01-09T17:15:55.111819Z","iopub.status.idle":"2026-01-09T17:15:55.129041Z","shell.execute_reply.started":"2026-01-09T17:15:55.111789Z","shell.execute_reply":"2026-01-09T17:15:55.127857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nimport gc\n\n# 1. Đọc file dưới dạng LazyFrame\nranker_lazy = pl.scan_parquet('/kaggle/input/hm-lgbm-rank-test/valid_ranker_lgbm.pqt')\nbinary_lazy = pl.scan_parquet('/kaggle/input/hm-result-tune/valid_binary_lgbm.pqt')\n\n# 2. Xử lý Ranker: Group by và gom thành list\n# Lưu ý: Dùng .group_by thay vì .groupby\nranker_agg = ranker_lazy.group_by(\"customer_id\").agg([\n    pl.col(\"prediction\").alias(\"lgbm_rank\")\n])\n\n# 3. Xử lý Binary: Group by và gom thành list\nbinary_agg = binary_lazy.group_by(\"customer_id\").agg([\n    pl.col(\"prediction\").alias(\"lgbm_binary\")\n])\n\n# 4. Join hai bảng đã gom nhóm trên customer_id\n# Vì đã gom nhóm nên mỗi customer_id là duy nhất, join sẽ cực nhẹ\nfinal_lazy = ranker_agg.join(binary_agg, on=\"customer_id\", how=\"inner\")\n\n# 5. Thực thi với engine streaming mới nhất\n# Thay vì streaming=True, dùng engine=\"streaming\" (hoặc \"old-streaming\" nếu bản polars quá mới)\ntry:\n    pred = final_lazy.collect(engine=\"streaming\")\nexcept Exception:\n    # Nếu engine mới chưa hỗ trợ một số operator, dùng mặc định\n    pred = final_lazy.collect()\n\n# 6. Giải phóng bộ nhớ\ndel ranker_lazy, binary_lazy, ranker_agg, binary_agg, final_lazy\ngc.collect()\n\nprint(pred.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:55.130531Z","iopub.execute_input":"2026-01-09T17:15:55.130903Z","iopub.status.idle":"2026-01-09T17:15:56.416683Z","shell.execute_reply.started":"2026-01-09T17:15:55.130847Z","shell.execute_reply":"2026-01-09T17:15:56.414875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_blend(row):\n    # Polars truyền dữ liệu vào dưới dạng dict khi dùng struct\n    # row = {\"lgbm_rank\": [...], \"lgbm_binary\": [...]}\n    return cust_blend(row, W=[0, 1])\n\n# Áp dụng hàm blend ngay trong Polars\npred = pred.with_columns([\n    pl.struct([\"lgbm_rank\", \"lgbm_binary\"])\n    .map_elements(apply_blend, return_dtype=pl.List(pl.Int64)) # hoặc pl.List(pl.Utf8) tùy ID của bạn\n    .alias(\"prediction\")\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:15:56.418241Z","iopub.execute_input":"2026-01-09T17:15:56.418669Z","iopub.status.idle":"2026-01-09T17:16:05.671868Z","shell.execute_reply.started":"2026-01-09T17:15:56.418617Z","shell.execute_reply":"2026-01-09T17:16:05.670860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"actual_lazy = pl.scan_parquet('/kaggle/input/h-and-m-recall-model-dataset/week1_label.pqt')\n\nactual_processed = (\n    actual_lazy\n    .select([\"customer_id\", \"article_id\"])\n    .rename({\"article_id\": \"actual\"})\n)\n\n# --- BƯỚC 2: JOIN VỚI DỰ ĐOÁN (PRED) ĐỂ ĐÁNH GIÁ ---\n# Lưu ý: 'pred' ở bước trước đang là Polars DataFrame\neval_df = (\n    actual_processed\n    .join(pred.lazy(), on=\"customer_id\", how=\"left\")\n    .with_columns(\n        pl.col(\"prediction\").fill_null(pl.lit([])) # Thay NaN bằng list rỗng\n    )\n    .collect() # Thực thi để lấy dữ liệu tính toán\n)\n\n# --- BƯỚC 3: TÍNH TOÁN MAP@12 ---\nactual_list = eval_df[\"actual\"].to_list()\npredicted_list = eval_df[\"prediction\"].to_list()\n\nmap_result = map_at_k(actual_list, predicted_list, k=12)\nprint(f\"Validation MAP@12: {map_result:.6f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:16:05.673235Z","iopub.execute_input":"2026-01-09T17:16:05.673799Z","iopub.status.idle":"2026-01-09T17:16:06.119939Z","shell.execute_reply.started":"2026-01-09T17:16:05.673761Z","shell.execute_reply":"2026-01-09T17:16:06.118552Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Result\n","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport gc\n\n# 1. Đọc file dưới dạng LazyFrame\nranker_lazy = pl.scan_parquet('/kaggle/input/hm-lgbm-rank-test/lgbm_rank_test.pqt')\nbinary_lazy = pl.scan_parquet('/kaggle/input/hm-result-tune/lgbm_binary_test.pqt')\n\n# 2. Xử lý Ranker: Group by và gom thành list\n# Lưu ý: Dùng .group_by thay vì .groupby\nranker_agg = ranker_lazy.group_by(\"customer_id\").agg([\n    pl.col(\"prediction\").alias(\"lgbm_rank\")\n])\n\n# 3. Xử lý Binary: Group by và gom thành list\nbinary_agg = binary_lazy.group_by(\"customer_id\").agg([\n    pl.col(\"prediction\").alias(\"lgbm_binary\")\n])\n\n# 4. Join hai bảng đã gom nhóm trên customer_id\n# Vì đã gom nhóm nên mỗi customer_id là duy nhất, join sẽ cực nhẹ\nfinal_lazy = ranker_agg.join(binary_agg, on=\"customer_id\", how=\"inner\")\n\n# 5. Thực thi với engine streaming mới nhất\n# Thay vì streaming=True, dùng engine=\"streaming\" (hoặc \"old-streaming\" nếu bản polars quá mới)\ntry:\n    pred = final_lazy.collect(engine=\"streaming\")\nexcept Exception:\n    # Nếu engine mới chưa hỗ trợ một số operator, dùng mặc định\n    pred = final_lazy.collect()\n\n# 6. Giải phóng bộ nhớ\ndel ranker_lazy, binary_lazy, ranker_agg, binary_agg, final_lazy\ngc.collect()\n\nprint(pred.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:16:06.121240Z","iopub.execute_input":"2026-01-09T17:16:06.121686Z","iopub.status.idle":"2026-01-09T17:16:27.314143Z","shell.execute_reply.started":"2026-01-09T17:16:06.121644Z","shell.execute_reply":"2026-01-09T17:16:27.313187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_blend(row):\n    # Polars truyền dữ liệu vào dưới dạng dict khi dùng struct\n    # row = {\"lgbm_rank\": [...], \"lgbm_binary\": [...]}\n    return cust_blend(row, W=[0, 1])\n\n# Áp dụng hàm blend ngay trong Polars\npred = pred.with_columns([\n    pl.struct([\"lgbm_rank\", \"lgbm_binary\"])\n    .map_elements(apply_blend, return_dtype=pl.List(pl.Int64)) # hoặc pl.List(pl.Utf8) tùy ID của bạn\n    .alias(\"prediction\")\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:16:27.315725Z","iopub.execute_input":"2026-01-09T17:16:27.316077Z","iopub.status.idle":"2026-01-09T17:18:42.065918Z","shell.execute_reply.started":"2026-01-09T17:16:27.316038Z","shell.execute_reply":"2026-01-09T17:18:42.064639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(pred.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:18:42.067354Z","iopub.execute_input":"2026-01-09T17:18:42.067743Z","iopub.status.idle":"2026-01-09T17:18:42.074059Z","shell.execute_reply.started":"2026-01-09T17:18:42.067706Z","shell.execute_reply":"2026-01-09T17:18:42.072520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nimport gc\n\n# 1. Load các file mapping\nidx2uid = pickle.load(open(\"/kaggle/input/h-and-m-csv-dataset/user_index2id.pkl\", \"rb\"))\nidx2iid = pickle.load(open(\"/kaggle/input/h-and-m-csv-dataset/item_index2id.pkl\", \"rb\"))\n# Chuyển dictionary sang Polars DataFrame để Join (nhanh và tiết kiệm RAM hơn .map)\ndf_idx2uid = pl.DataFrame({\n    \"customer_id\": list(idx2uid.keys()),\n    \"customer_id_str\": list(idx2uid.values())\n}).with_columns(pl.col(\"customer_id\").cast(pl.Int64))\n\n# 2. Xử lý hàm parse ID bài viết (Article ID)\n# Thay vì dùng tqdm + apply, ta sử dụng map_elements của Polars nhưng chỉ cho việc mapping ID\ndef map_article_ids(indices):\n    if indices is None or len(indices) == 0:\n        return \"0115650031\" # Default item ID (giả sử là string 10 số)\n    # Lấy 12 index đầu tiên, map sang ID gốc và thêm số '0' ở đầu\n    return \" \".join([f\"0{idx2iid[i]}\" for i in indices[:NUM]])\n\n# 3. Chuyển đổi 'pred' hiện tại (đang là Polars DataFrame từ bước trước)\npred = pred.with_columns([\n    pl.col(\"prediction\").map_elements(map_article_ids, return_dtype=pl.Utf8)\n])\n\n# 4. Đọc Sample Submission bằng Polars\nsub = pl.scan_csv('/kaggle/input/h-and-m-csv-dataset/sample_submission.csv')\n\n# 5. Map customer_id của submission sang index (dùng uid2idx)\nuid2idx = pickle.load(open(\"/kaggle/input/mapping/index_id_map/user_id2index.pkl\", \"rb\"))\ndf_uid2idx = pl.DataFrame({\n    \"customer_id_str\": list(uid2idx.keys()),\n    \"customer_id\": list(uid2idx.values())\n})\n\n# 6. Thực hiện quy trình Join\n# Sub (str) -> Map to Index -> Join with Pred -> Fill Null -> Map back to String ID\nfinal_sub = (\n    sub.select(pl.col(\"customer_id\").alias(\"customer_id_str\"))\n    .join(df_uid2idx.lazy(), on=\"customer_id_str\", how=\"left\")\n    .join(pred.lazy(), on=\"customer_id\", how=\"left\")\n    .with_columns(\n        pl.col(\"prediction\").fill_null(\"0115650031\") # Xử lý các khách hàng không có dự đoán\n    )\n    .select([\n        pl.col(\"customer_id_str\").alias(\"customer_id\"),\n        pl.col(\"prediction\")\n    ])\n)\n\n# 7. Thực thi và xuất file\n# Dùng collect() ở bước cuối cùng để thực hiện toàn bộ pipeline\nsubmission_df = final_sub.collect(engine=\"streaming\")\nsubmission_df.write_csv(\"output.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:18:42.075678Z","iopub.execute_input":"2026-01-09T17:18:42.076074Z","iopub.status.idle":"2026-01-09T17:19:04.658888Z","shell.execute_reply.started":"2026-01-09T17:18:42.076040Z","shell.execute_reply":"2026-01-09T17:19:04.657112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.660394Z","iopub.execute_input":"2026-01-09T17:19:04.660837Z","iopub.status.idle":"2026-01-09T17:19:04.669244Z","shell.execute_reply.started":"2026-01-09T17:19:04.660787Z","shell.execute_reply":"2026-01-09T17:19:04.668163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ranker_lightgbm_test= pd.read_parquet('/kaggle/input/hm-ranker-lightgbm-test/lgbm_rank_test.pqt')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.670469Z","iopub.execute_input":"2026-01-09T17:19:04.670803Z","iopub.status.idle":"2026-01-09T17:19:04.697416Z","shell.execute_reply.started":"2026-01-09T17:19:04.670770Z","shell.execute_reply":"2026-01-09T17:19:04.696246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ranker_lightgbm_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.699034Z","iopub.execute_input":"2026-01-09T17:19:04.699360Z","iopub.status.idle":"2026-01-09T17:19:04.717343Z","shell.execute_reply.started":"2026-01-09T17:19:04.699325Z","shell.execute_reply":"2026-01-09T17:19:04.716075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ranker_lightgbm_test=ranker_lightgbm_test.sort_values(by=['customer_id', 'prob'], ascending=False).reset_index(drop=True)\n# ranker_lightgbm_test=ranker_lightgbm_test.groupby('customer_id')['prediction'].apply(list).reset_index()\n# ranker_lightgbm_test = ranker_lightgbm_test.rename(columns={'prediction': 'lgbm_rank'})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.718601Z","iopub.execute_input":"2026-01-09T17:19:04.718974Z","iopub.status.idle":"2026-01-09T17:19:04.736417Z","shell.execute_reply.started":"2026-01-09T17:19:04.718935Z","shell.execute_reply":"2026-01-09T17:19:04.735340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(ranker_lightgbm_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.737795Z","iopub.execute_input":"2026-01-09T17:19:04.738196Z","iopub.status.idle":"2026-01-09T17:19:04.755037Z","shell.execute_reply.started":"2026-01-09T17:19:04.738157Z","shell.execute_reply":"2026-01-09T17:19:04.754019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# binary_lightgbm_test= pd.read_parquet('/kaggle/input/h-and-m-binary-lightgbm-test/lgbm_binary_test.pqt')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.756270Z","iopub.execute_input":"2026-01-09T17:19:04.756836Z","iopub.status.idle":"2026-01-09T17:19:04.775560Z","shell.execute_reply.started":"2026-01-09T17:19:04.756791Z","shell.execute_reply":"2026-01-09T17:19:04.774290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# binary_lightgbm_test=binary_lightgbm_test.sort_values(by=['customer_id', 'prob'], ascending=False).reset_index(drop=True)\n# binary_lightgbm_test=binary_lightgbm_test.groupby('customer_id')['prediction'].apply(list).reset_index()\n# binary_lightgbm_test = binary_lightgbm_test.rename(columns={'prediction': 'lgbm_binary'})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.776859Z","iopub.execute_input":"2026-01-09T17:19:04.777221Z","iopub.status.idle":"2026-01-09T17:19:04.792968Z","shell.execute_reply.started":"2026-01-09T17:19:04.777182Z","shell.execute_reply":"2026-01-09T17:19:04.791688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# binary_lightgbm_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.797996Z","iopub.execute_input":"2026-01-09T17:19:04.798359Z","iopub.status.idle":"2026-01-09T17:19:04.811561Z","shell.execute_reply.started":"2026-01-09T17:19:04.798322Z","shell.execute_reply":"2026-01-09T17:19:04.810616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred = pd.merge(ranker_lightgbm_test, binary_lightgbm_test, on='customer_id')\n# binary = binary_lightgbm_test.set_index('customer_id')\n\n# chunks = []\n# chunk_size = 1_000_000  # tùy RAM\n\n# for i in range(0, len(ranker_lightgbm_test), chunk_size):\n#     chunk = ranker_lightgbm_test.iloc[i:i+chunk_size]\n#     chunk = chunk.set_index('customer_id').join(binary, how='inner')\n#     chunks.append(chunk.reset_index())\n\n# pred = pd.concat(chunks, ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.812739Z","iopub.execute_input":"2026-01-09T17:19:04.813058Z","iopub.status.idle":"2026-01-09T17:19:04.831413Z","shell.execute_reply.started":"2026-01-09T17:19:04.813030Z","shell.execute_reply":"2026-01-09T17:19:04.830271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.832842Z","iopub.execute_input":"2026-01-09T17:19:04.833543Z","iopub.status.idle":"2026-01-09T17:19:04.851149Z","shell.execute_reply.started":"2026-01-09T17:19:04.833346Z","shell.execute_reply":"2026-01-09T17:19:04.849732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# del binary_lightgbm_test, ranker_lightgbm_test\n# gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.852471Z","iopub.execute_input":"2026-01-09T17:19:04.852767Z","iopub.status.idle":"2026-01-09T17:19:04.869762Z","shell.execute_reply.started":"2026-01-09T17:19:04.852738Z","shell.execute_reply":"2026-01-09T17:19:04.868573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(len(pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.871033Z","iopub.execute_input":"2026-01-09T17:19:04.871421Z","iopub.status.idle":"2026-01-09T17:19:04.888660Z","shell.execute_reply.started":"2026-01-09T17:19:04.871355Z","shell.execute_reply":"2026-01-09T17:19:04.887626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred['prediction'] = pred.apply(lambda row: cust_blend(row, W=[1.3, 1]), axis=1)\n# print(\"The 'cust_blend' function has been applied to create 'blended_recommendations' column.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.890260Z","iopub.execute_input":"2026-01-09T17:19:04.890623Z","iopub.status.idle":"2026-01-09T17:19:04.909363Z","shell.execute_reply.started":"2026-01-09T17:19:04.890593Z","shell.execute_reply":"2026-01-09T17:19:04.908386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.910672Z","iopub.execute_input":"2026-01-09T17:19:04.911024Z","iopub.status.idle":"2026-01-09T17:19:04.929553Z","shell.execute_reply.started":"2026-01-09T17:19:04.910986Z","shell.execute_reply":"2026-01-09T17:19:04.928125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.931603Z","iopub.execute_input":"2026-01-09T17:19:04.931996Z","iopub.status.idle":"2026-01-09T17:19:04.948981Z","shell.execute_reply.started":"2026-01-09T17:19:04.931959Z","shell.execute_reply":"2026-01-09T17:19:04.947906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# idx2uid = pickle.load(open(\"/kaggle/input/h-and-m-csv-dataset/user_index2id.pkl\", \"rb\"))\n# idx2iid = pickle.load(open(\"/kaggle/input/h-and-m-csv-dataset/item_index2id.pkl\", \"rb\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.950790Z","iopub.execute_input":"2026-01-09T17:19:04.951593Z","iopub.status.idle":"2026-01-09T17:19:04.969367Z","shell.execute_reply.started":"2026-01-09T17:19:04.951560Z","shell.execute_reply":"2026-01-09T17:19:04.968253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def parse(x):\n#     return ' '.join(\n#         '0' + str(idx2iid[i]) for i in x[:12]\n#     )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.970801Z","iopub.execute_input":"2026-01-09T17:19:04.971219Z","iopub.status.idle":"2026-01-09T17:19:04.988852Z","shell.execute_reply.started":"2026-01-09T17:19:04.971178Z","shell.execute_reply":"2026-01-09T17:19:04.987779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# tqdm.pandas(desc=\"Processing prediction\")\n\n# pred['prediction'] = [\n#     parse(x)\n#     for x in tqdm(pred['prediction'], total=len(pred))\n# ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:04.990444Z","iopub.execute_input":"2026-01-09T17:19:04.990943Z","iopub.status.idle":"2026-01-09T17:19:05.014221Z","shell.execute_reply.started":"2026-01-09T17:19:04.990910Z","shell.execute_reply":"2026-01-09T17:19:05.012936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# uid2idx = pickle.load(open(\"/kaggle/input/mapping/index_id_map/user_id2index.pkl\", \"rb\"))\n# submission = pd.read_csv(f'/kaggle/input/h-and-m-csv-dataset/sample_submission.csv')\n# submission['customer_id'] = submission['customer_id'].map(uid2idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.015667Z","iopub.execute_input":"2026-01-09T17:19:05.016092Z","iopub.status.idle":"2026-01-09T17:19:05.034012Z","shell.execute_reply.started":"2026-01-09T17:19:05.016060Z","shell.execute_reply":"2026-01-09T17:19:05.033039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# del submission['prediction']\n# submission = submission.merge(pred, on='customer_id', how='left')\n# submission['customer_id'] = submission['customer_id'].map(idx2uid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.035422Z","iopub.execute_input":"2026-01-09T17:19:05.035853Z","iopub.status.idle":"2026-01-09T17:19:05.053498Z","shell.execute_reply.started":"2026-01-09T17:19:05.035810Z","shell.execute_reply":"2026-01-09T17:19:05.052532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission = submission[['customer_id', 'prediction']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.054545Z","iopub.execute_input":"2026-01-09T17:19:05.054846Z","iopub.status.idle":"2026-01-09T17:19:05.072857Z","shell.execute_reply.started":"2026-01-09T17:19:05.054819Z","shell.execute_reply":"2026-01-09T17:19:05.071753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(submission['prediction'].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.074198Z","iopub.execute_input":"2026-01-09T17:19:05.074585Z","iopub.status.idle":"2026-01-09T17:19:05.090667Z","shell.execute_reply.started":"2026-01-09T17:19:05.074541Z","shell.execute_reply":"2026-01-09T17:19:05.089663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission['prediction'] = submission['prediction'].apply(\n#     lambda x: [11565003] if (x is None or (isinstance(x, float) and pd.isna(x)) or x == []) else x\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.092111Z","iopub.execute_input":"2026-01-09T17:19:05.092686Z","iopub.status.idle":"2026-01-09T17:19:05.108356Z","shell.execute_reply.started":"2026-01-09T17:19:05.092641Z","shell.execute_reply":"2026-01-09T17:19:05.107178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(submission['prediction'].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.109645Z","iopub.execute_input":"2026-01-09T17:19:05.109957Z","iopub.status.idle":"2026-01-09T17:19:05.126892Z","shell.execute_reply.started":"2026-01-09T17:19:05.109930Z","shell.execute_reply":"2026-01-09T17:19:05.125854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.128201Z","iopub.execute_input":"2026-01-09T17:19:05.128602Z","iopub.status.idle":"2026-01-09T17:19:05.145709Z","shell.execute_reply.started":"2026-01-09T17:19:05.128562Z","shell.execute_reply":"2026-01-09T17:19:05.144706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# len(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.147821Z","iopub.execute_input":"2026-01-09T17:19:05.148124Z","iopub.status.idle":"2026-01-09T17:19:05.166037Z","shell.execute_reply.started":"2026-01-09T17:19:05.148097Z","shell.execute_reply":"2026-01-09T17:19:05.164833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission.to_csv('ouput.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.167215Z","iopub.execute_input":"2026-01-09T17:19:05.167563Z","iopub.status.idle":"2026-01-09T17:19:05.186215Z","shell.execute_reply.started":"2026-01-09T17:19:05.167535Z","shell.execute_reply":"2026-01-09T17:19:05.185029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data = pd.read_csv(f'/kaggle/input/h-and-m-feature-engineering/enriched_week1.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.187451Z","iopub.execute_input":"2026-01-09T17:19:05.187769Z","iopub.status.idle":"2026-01-09T17:19:05.206608Z","shell.execute_reply.started":"2026-01-09T17:19:05.187741Z","shell.execute_reply":"2026-01-09T17:19:05.205346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data = data.sort_values(by=['customer_id'], ascending=False).reset_index(drop=True)\n# data = data[data['label'] == 1][['customer_id','article_id','label']]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.208058Z","iopub.execute_input":"2026-01-09T17:19:05.208431Z","iopub.status.idle":"2026-01-09T17:19:05.225726Z","shell.execute_reply.started":"2026-01-09T17:19:05.208391Z","shell.execute_reply":"2026-01-09T17:19:05.224519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# actual_purchases = data.groupby('customer_id')['article_id'].apply(list).reset_index()\n\n# # Merge actual purchases with the blended recommendations\n# merged_evaluation_data = pd.merge(res[['customer_id', 'blended_recommendations']], actual_purchases, on='customer_id', how='left')\n\n# # Fill NaN values in 'article_id' (for customers with no actual purchases in the evaluation set) with empty lists\n# merged_evaluation_data['article_id'] = merged_evaluation_data['article_id'].apply(lambda x: x if isinstance(x, list) else [])\n\n# # Prepare lists for map_at_k function\n# actual_list = merged_evaluation_data['article_id'].tolist()\n# predicted_list = merged_evaluation_data['blended_recommendations'].tolist()\n\n# # Calculate MAP@12\n# result = map_at_k(actual_list, predicted_list, k=12)\n# print(f\"MAP@12: {result}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-09T17:19:05.227121Z","iopub.execute_input":"2026-01-09T17:19:05.227472Z","iopub.status.idle":"2026-01-09T17:19:05.244609Z","shell.execute_reply.started":"2026-01-09T17:19:05.227442Z","shell.execute_reply":"2026-01-09T17:19:05.243650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}