{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":14112508,"sourceType":"datasetVersion","datasetId":8989713},{"sourceId":14241442,"sourceType":"datasetVersion","datasetId":9085906},{"sourceId":14258748,"sourceType":"datasetVersion","datasetId":9098345},{"sourceId":14317728,"sourceType":"datasetVersion","datasetId":9126526},{"sourceId":14387225,"sourceType":"datasetVersion","datasetId":9185659}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport gc\nimport re\nfrom datetime import timedelta\nimport traceback\nimport pyarrow as pa\nimport pyarrow.parquet as pq\nimport pickle\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:02:13.714318Z","iopub.execute_input":"2026-01-04T05:02:13.714716Z","iopub.status.idle":"2026-01-04T05:02:14.187075Z","shell.execute_reply.started":"2026-01-04T05:02:13.714689Z","shell.execute_reply":"2026-01-04T05:02:14.186065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CẤU HÌNH CONFIG ---\nclass Config:\n    # 1. Đường dẫn dataset gốc\n    RAW_DATA_DIR = '/kaggle/input/h-and-m-personalized-fashion-recommendations/'\n    \n    # 2. Đường dẫn dataset Candidate\n    CANDIDATE_DIR = '/kaggle/input/itemgrouptimehistory/'\n    \n    # 3. Thư mục xuất file kết quả\n    OUTPUT_DIR = '/kaggle/working/enriched_data/'\n    \n    # 4. Các Feature CẦN LOẠI BỎ\n    DROP_FEATURES = ['dssm_similarity', 'yt_similarity', 'wv_similarity', 'label']\n    \n    # 5. Kích thước chunk\n    CHUNK_SIZE = 200_000 # Giảm xuống 200k cho an toàn RAM\n\n    # 6. Thư mục các labels\n    LABEL_DIR = \"/kaggle/input/candidates-dataset/\"\n\n    MAPPING_DIR = \"/kaggle/input/mapping/index_id_map/\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:02:14.188326Z","iopub.execute_input":"2026-01-04T05:02:14.188718Z","iopub.status.idle":"2026-01-04T05:02:14.195579Z","shell.execute_reply.started":"2026-01-04T05:02:14.188697Z","shell.execute_reply":"2026-01-04T05:02:14.194301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# 1. UTILS\n# =============================================================================\ndef reduce_mem_usage(df):\n    \"\"\"Giảm dung lượng RAM, bỏ qua cột datetime\"\"\"\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object and not np.issubdtype(col_type, np.datetime64):\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                else:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float32)\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:02:14.196708Z","iopub.execute_input":"2026-01-04T05:02:14.197030Z","iopub.status.idle":"2026-01-04T05:02:14.225224Z","shell.execute_reply.started":"2026-01-04T05:02:14.197000Z","shell.execute_reply":"2026-01-04T05:02:14.223965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from typing import Tuple\ndef calc_valid_date(week_num: int, last_date: str = \"2020-09-29\") -> Tuple[str]:\n    \"\"\"Calculate start and end date of a given week number.\n\n    Parameters\n    ----------\n    week_num : int\n        Week number.\n    last_date : str, optional\n        The last day, by default ``\"2020-09-22\"``.\n\n    Returns\n    -------\n    Tuple[str]\n        Start and end date of the given week number.\n    \"\"\"\n    end_date = pd.to_datetime(last_date) - pd.Timedelta(days=7 * week_num - 1)\n    start_date = end_date - pd.Timedelta(days=7)\n\n    end_date = end_date.strftime(\"%Y-%m-%d\")\n    start_date = start_date.strftime(\"%Y-%m-%d\")\n    return start_date, end_date\n\ndef calc_valid_week(dat_t: str, last_date: str = '2020-09-29') -> int:\n    week_num = (pd.to_datetime(last_date) - pd.to_datetime(dat_t)) // 7\n    return week_num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:02:14.227888Z","iopub.execute_input":"2026-01-04T05:02:14.228805Z","iopub.status.idle":"2026-01-04T05:02:14.252446Z","shell.execute_reply.started":"2026-01-04T05:02:14.228763Z","shell.execute_reply":"2026-01-04T05:02:14.251273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_path = Config.RAW_DATA_DIR\npqt_path = Config.CANDIDATE_DIR\noutput_dir = Config.OUTPUT_DIR\nmapping_dir = Config.MAPPING_DIR\nprint(\"\\n=== [1] PREPARING RESOURCES ===\")\n\n# --- A. LOAD CUSTOMERS ---\nprint(\"-> Loading Customers...\")\ncustomers = pd.read_csv(raw_path + 'customers.csv')\nwith open(mapping_dir + 'user_id2index.pkl', \"rb\") as f:\n    cust_id_map = pickle.load(f)\n\ncustomers['customer_id'] = customers.index.astype('int32')\ncustomers['age'] = customers['age'].fillna(customers['age'].mean()).astype(np.int8)\ndel customers; gc.collect()\n\n# --- B. LOAD ARTICLES ---\nprint(\"-> Loading Articles...\")\narticles = pd.read_csv(raw_path + 'articles.csv', dtype={'article_id': str})\nwith open(mapping_dir + 'item_id2index.pkl', \"rb\") as f:\n    article_id_map = pickle.load(f)\ndel articles; gc.collect()\n\n# --- C. LOAD TRANSACTIONS ---\nprint(\"-> Loading Transactions (Last 5 weeks)...\")\ndf_trans = pd.read_csv(raw_path + 'transactions_train.csv')\n\nprint(\"   Mapping IDs...\")\ndf_trans['customer_id'] = df_trans['customer_id'].map(cust_id_map).fillna(-1).astype('int32')\ndf_trans['article_id'] = df_trans['article_id'].map(article_id_map).fillna(-1).astype('int32')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:02:14.253534Z","iopub.execute_input":"2026-01-04T05:02:14.253851Z","iopub.status.idle":"2026-01-04T05:03:39.490427Z","shell.execute_reply.started":"2026-01-04T05:02:14.253827Z","shell.execute_reply":"2026-01-04T05:03:39.488521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"last_date = pd.to_datetime('2020-09-29')\n\ndf_trans['week'] = (\n    last_date - pd.to_datetime(df_trans['t_dat'])\n).dt.days // 7\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:03:39.491869Z","iopub.execute_input":"2026-01-04T05:03:39.492316Z","iopub.status.idle":"2026-01-04T05:03:44.187670Z","shell.execute_reply.started":"2026-01-04T05:03:39.492279Z","shell.execute_reply":"2026-01-04T05:03:44.186530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:03:44.189021Z","iopub.execute_input":"2026-01-04T05:03:44.189363Z","iopub.status.idle":"2026-01-04T05:03:44.222807Z","shell.execute_reply.started":"2026-01-04T05:03:44.189334Z","shell.execute_reply":"2026-01-04T05:03:44.221492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_rolling_avg_week(\n    df,\n    group_col,\n    week_col='week',\n    value_col='user_avg_spend_1w',\n    windows=(2, 3, 4, 6),\n    prefix='avg'\n):\n    df = df.copy()\n\n    # QUAN TRỌNG: week nhỏ = gần → sort DESC\n    df = df.sort_values([group_col, week_col], ascending=[True, False])\n\n    for w in windows:\n        col = f'{prefix}_{w}w'\n        df[col] = (\n            df\n            .groupby(group_col)[value_col]\n            .rolling(window=w, min_periods=1)\n            .mean()\n            .reset_index(level=0, drop=True)\n        )\n\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:03:44.223877Z","iopub.execute_input":"2026-01-04T05:03:44.224201Z","iopub.status.idle":"2026-01-04T05:03:44.231181Z","shell.execute_reply.started":"2026-01-04T05:03:44.224174Z","shell.execute_reply":"2026-01-04T05:03:44.229798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(\"-> Calculating Helpers...\")\n\navg_price_week = df_trans.groupby('week')['price'].mean().reset_index(name='avg_price_1w').sort_values('week', ascending=False) \n\nfor w in [2, 4]:\n    avg_price_week[f'avg_price_{w}w'] = (\n        avg_price_week['avg_price_1w']\n        .rolling(window=w, min_periods=1)\n        .mean()\n    )\n\n\n# df_trans = df_trans.merge(\n#     avg_price_week,\n#     on=['week'],\n#     how='left'\n# )\n\nuser_avg_spend_1w = (\n    df_trans\n    .groupby(['customer_id', 'week'], as_index=False)\n    .agg(user_avg_spend_1w=('price', 'mean'))\n    .sort_values(['customer_id', 'week'])\n)\n\n\n\nitem_avg_price_1w = (\n    df_trans\n    .groupby(['article_id', 'week'], as_index=False)\n    .agg(item_avg_price_1w=('price', 'mean'))\n    .sort_values(['article_id', 'week'])\n)\n\nuser_week_price = add_rolling_avg_week(\n    user_avg_spend_1w,\n    group_col='customer_id',\n    windows=(2, 4),\n    prefix='user_avg_spend'\n)\n\nuser_week_price = user_week_price.merge(avg_price_week, on=['week'], how='left')\n\nitem_week_price = add_rolling_avg_week(\n    item_avg_price_1w,\n    group_col='article_id',\n    windows=(2, 4),\n    value_col='item_avg_price_1w',\n    prefix='item_avg_price'\n)\n\nitem_week_price = item_week_price.merge(avg_price_week, on=['week'], how='left')\n\n\n# df_trans = df_trans.merge(\n#     user_week_price,\n#     on=['customer_id', 'week'],\n#     how='left'\n# )\n\n# df_trans = df_trans.merge(\n#     item_week_price,\n#     on=['article_id', 'week'],\n#     how='left'\n# ) \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:03:44.232745Z","iopub.execute_input":"2026-01-04T05:03:44.233025Z","iopub.status.idle":"2026-01-04T05:05:16.417684Z","shell.execute_reply.started":"2026-01-04T05:03:44.233004Z","shell.execute_reply":"2026-01-04T05:05:16.416581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in [1, 2, 4]:\n    user_week_price[f'diff_user_avg_spend_{i}w'] = user_week_price[f'user_avg_spend_{i}w'] - user_week_price[f'avg_price_{i}w']\n    item_week_price[f'diff_item_avg_spend_{i}w'] = item_week_price[f'item_avg_price_{i}w'] - item_week_price[f'avg_price_{i}w']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:05:16.420672Z","iopub.execute_input":"2026-01-04T05:05:16.420972Z","iopub.status.idle":"2026-01-04T05:05:16.544272Z","shell.execute_reply.started":"2026-01-04T05:05:16.420951Z","shell.execute_reply":"2026-01-04T05:05:16.543264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans = reduce_mem_usage(df_trans)\nuser_week_price = reduce_mem_usage(user_week_price)\nitem_week_price = reduce_mem_usage(item_week_price)\navg_price_week = reduce_mem_usage(avg_price_week)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:05:16.545607Z","iopub.execute_input":"2026-01-04T05:05:16.545905Z","iopub.status.idle":"2026-01-04T05:05:17.915909Z","shell.execute_reply.started":"2026-01-04T05:05:16.545883Z","shell.execute_reply":"2026-01-04T05:05:17.914526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans[\"t_dat\"] = pd.to_datetime(df_trans[\"t_dat\"])\ndf_trans['year'] = df_trans['t_dat'].dt.year\ndf_period = df_trans[\n    (df_trans['t_dat'].dt.month > 1) |\n    ((df_trans['t_dat'].dt.month == 1) & (df_trans['t_dat'].dt.day >= 1))\n]\n\ndf_period = df_period[\n    (df_period['t_dat'].dt.month < 9)\n]\nyearly_avg = (\n    df_period\n    .groupby('year', as_index=False)\n    .agg(\n        avg_price=('price', 'mean'),\n        cnt=('price', 'count')\n    )\n    .sort_values('year')\n)\navg_2019 = yearly_avg.loc[yearly_avg['year'] == 2019, 'avg_price'].values[0]\navg_2020 = yearly_avg.loc[yearly_avg['year'] == 2020, 'avg_price'].values[0]\n\ndiff_2020_2019 = avg_2020 - avg_2019\n\ndiff_2020_2019\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:05:17.917313Z","iopub.execute_input":"2026-01-04T05:05:17.917626Z","iopub.status.idle":"2026-01-04T05:05:28.561162Z","shell.execute_reply.started":"2026-01-04T05:05:17.917592Z","shell.execute_reply":"2026-01-04T05:05:28.560329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport re\n\ndef get_files_of_week(base_dir, week):\n    \"\"\"\n    base_dir: thư mục chứa file (ví dụ: \"/kaggle/input/itemgrouptimehistory\")\n    week: tuần cần lấy (int), ví dụ: 3\n    \"\"\"\n\n    files = [\n        os.path.join(base_dir, f)\n        for f in os.listdir(base_dir)\n        if os.path.isfile(os.path.join(base_dir, f))\n    ]\n\n    pattern = re.compile(r'week(\\d+)')\n\n    result = []\n\n    for f in files:\n        m = pattern.search(f)\n        if m and int(m.group(1)) == week:\n            result.append(f)\n\n    return result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:05:28.562145Z","iopub.execute_input":"2026-01-04T05:05:28.562451Z","iopub.status.idle":"2026-01-04T05:05:28.568770Z","shell.execute_reply.started":"2026-01-04T05:05:28.562430Z","shell.execute_reply":"2026-01-04T05:05:28.567718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\n\ndef evaluate_rule(rule_name, candidates, results, valid_exploded, week, alpha):\n    print(f\"\\n====== TEST RULE: {rule_name} (week {week}) (alpha {alpha})======\")\n\n    # đảm bảo unique candidate\n    candidates = (\n        candidates[['customer_id', 'article_id']]\n        .drop_duplicates()\n        .reset_index(drop=True)\n    )\n\n    # ---- CALCULATE PRECISION & RECALL ----\n    matched = candidates.merge(\n        valid_exploded[['customer_id','article_id']],\n        on=['customer_id','article_id'],\n        how='inner'\n    )\n\n    num_matched = len(matched)\n    total_valid = len(valid_exploded)\n    total_candidates = len(candidates)\n\n    recall = num_matched / total_valid if total_valid > 0 else 0\n    precision = num_matched / total_candidates if total_candidates > 0 else 0\n    f2 = (1+2**2)*precision*recall/(2**2*precision+recall)\n    new_result = {\n        \"num_candidates\": total_candidates,\n        \"num_matched\": num_matched,\n        \"precision\": precision,\n        \"recall\": recall,\n        \"f2-score\": f2\n    }\n\n    rows = results[results[\"rule\"] == rule_name]\n    precision = rows[\"precision\"].iloc[0]\n    recall = rows[\"recall\"].iloc[0]\n    num_matched = rows[\"num_matched\"].iloc[0]\n    num_candidates = rows[\"num_candidates\"].iloc[0]\n    f2 = (1+2**2)*precision*recall/(2**2*precision+recall)\n\n    old_result = {\n        \"num_candidates\": num_candidates,\n        \"num_matched\": num_matched,\n        \"precision\": precision,\n        \"recall\": recall,\n        \"f2-score\": f2\n    }\n    \n    print(f\"Old result: {old_result}\")\n    print(f\"New result: {new_result}\")\n    return new_result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:17:00.852352Z","iopub.execute_input":"2026-01-04T05:17:00.852707Z","iopub.status.idle":"2026-01-04T05:17:00.863320Z","shell.execute_reply.started":"2026-01-04T05:17:00.852682Z","shell.execute_reply":"2026-01-04T05:17:00.862346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def age_filer(candidates: pd.DataFrame, \n              avg_price_week: pd.DataFrame, \n              item_week_price: pd.DataFrame, \n              user_week_price: pd.DataFrame, \n              df_trans: pd.DataFrame, \n              week: int,\n              alpha: float,\n              std: float)->pd.DataFrame:\n    candidates = candidates[['customer_id', 'article_id']]\n    candidates['week'] = week\n    candidates = candidates.merge(avg_price_week, on=['week'], how='left')\n    candidates = candidates.merge(\n        user_week_price.drop(\n            columns=['avg_price_1w', 'avg_price_2w', 'avg_price_4w'],\n            errors='ignore'\n        ),\n        on=['week', 'customer_id'],\n        how='left'\n    )\n    \n    candidates = candidates.merge(\n        item_week_price.drop(\n            columns=['avg_price_1w', 'avg_price_2w', 'avg_price_4w'],\n            errors='ignore'\n        ),\n        on=['week', 'article_id'],\n        how='left'\n    )\n    \n    for j in [1, 2, 4]:\n        last_item_price = (\n            item_week_price[item_week_price[f\"diff_item_avg_spend_{j}w\"].notna()]\n            .sort_values(\"week\")\n            .groupby(\"article_id\")[f\"diff_item_avg_spend_{j}w\"]\n            .first()\n        )\n    \n        candidates[f\"diff_item_avg_spend_{j}w\"] = candidates[f\"diff_item_avg_spend_{j}w\"].fillna(\n            candidates[\"article_id\"].map(last_item_price)\n        )\n    \n        candidates[f\"item_avg_price_{j}w\"] = np.where(\n            candidates[f\"item_avg_price_{j}w\"].isna(),\n            candidates[f\"diff_item_avg_spend_{j}w\"] + candidates[f'avg_price_{j}w'],   # biểu thức trong cùng hàng\n            candidates[f\"item_avg_price_{j}w\"]\n        )\n    \n        last_user_price = (\n            user_week_price[user_week_price[f\"diff_user_avg_spend_{j}w\"].notna()]\n            .sort_values(\"week\")\n            .groupby(\"customer_id\")[f\"diff_user_avg_spend_{j}w\"]\n            .first()\n        )\n    \n        candidates[f\"diff_user_avg_spend_{j}w\"] = candidates[f\"diff_user_avg_spend_{j}w\"].fillna(\n            candidates[\"customer_id\"].map(last_user_price)\n        )\n    \n        candidates[f\"user_avg_spend_{j}w\"] = np.where(\n            candidates[f\"user_avg_spend_{j}w\"].isna(),\n            candidates[f\"diff_user_avg_spend_{j}w\"] + candidates[f'avg_price_{j}w'],   # biểu thức trong cùng hàng\n            candidates[f\"user_avg_spend_{j}w\"]\n        )\n        del last_item_price, last_user_price\n        gc.collect()\n        \n    item_last_buy_week = (\n            df_trans[df_trans[\"week\"] > week]\n            .groupby(\"article_id\", as_index=False)[\"week\"]\n            .min()\n            .rename(columns={\"week\": \"item_last_buy_week\"})\n        )\n    item_last_buy_week['week_nums_last_item'] = item_last_buy_week['item_last_buy_week'] - week\n\n    candidates = candidates.merge(\n        item_last_buy_week[[\"article_id\", \"week_nums_last_item\"]],\n        on=\"article_id\",\n        how=\"left\"\n    )\n    del item_last_buy_week, \n    candidates = candidates.drop(columns=[c for c in candidates.columns if c.startswith(\"diff\")])\n    candidates = candidates.drop(columns=['week'])\n    tmp = pd.DataFrame(index=candidates.index)\n    tmp[\"lower\"] = candidates[\"user_avg_spend_1w\"] - alpha * std\n    tmp[\"upper\"] = candidates[\"user_avg_spend_1w\"] + alpha * std\n    mask = False\n\n    mask |= (candidates[\"item_avg_price_1w\"] >= tmp[\"lower\"]) & \\\n        (candidates[\"item_avg_price_1w\"] <= tmp[\"upper\"])\n    \n    candidates = candidates.loc[mask, [\"customer_id\", \"article_id\"]].drop_duplicates()\n    return candidates","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:38:08.509692Z","iopub.execute_input":"2026-01-04T07:38:08.510002Z","iopub.status.idle":"2026-01-04T07:38:08.525155Z","shell.execute_reply.started":"2026-01-04T07:38:08.509978Z","shell.execute_reply":"2026-01-04T07:38:08.523901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# candidates = pd.read_parquet(\"/kaggle/input/itemgrouptimehistory/week1_IGTH_15_10_product_type_no_candidates.pqt\", columns=['customer_id', 'article_id'])\n# age_filer(candidates,\n#           avg_price_week, \n#           item_week_price, \n#           user_week_price, \n#           df_trans, \n#           week=1,\n#           alpha=2,\n#           std=0.01147336920442122)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:38:11.035037Z","iopub.execute_input":"2026-01-04T07:38:11.035337Z","iopub.status.idle":"2026-01-04T07:38:35.243608Z","shell.execute_reply.started":"2026-01-04T07:38:11.035314Z","shell.execute_reply":"2026-01-04T07:38:35.242675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"avg_price_week.to_csv(\"avg_price_week.csv\", index=False)\n\nuser_week_price.to_csv(\"user_week_price.csv\", index=False)\n\nitem_week_price.to_csv(\"item_week_price.csv\", index=False)\n\ndf_trans.to_csv(\"df_trans.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:11:30.956023Z","iopub.execute_input":"2026-01-04T07:11:30.956432Z","iopub.status.idle":"2026-01-04T07:16:08.882138Z","shell.execute_reply.started":"2026-01-04T07:11:30.956407Z","shell.execute_reply":"2026-01-04T07:16:08.880154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import numpy as np\n# import pyarrow.dataset as ds\n# import pyarrow as pa\n# import pyarrow.parquet as pq\n# import os\n\n# std = 0.01147336920442122\n# alphas = [1, 2, 3, 4]\n# results = pd.read_csv(f\"{Config.CANDIDATE_DIR}item_group_time_history_results.csv\")\n# new_results = []\n\n# for i in range(1, 6):\n#     links = get_files_of_week(base_dir=Config.CANDIDATE_DIR, week=i)\n#     valid = pd.read_parquet(f\"{Config.LABEL_DIR}week{i}_label.pqt\")\n#     valid_exploded = valid.explode('article_id')\n#     customer_list = valid[\"customer_id\"].values\n#     for link in links:\n#         rule_name = os.path.splitext(os.path.basename(link))[0]\n#         rule_name = rule_name.replace(\"_candidates\",\"\")\n#         week_df = pd.read_parquet(link, columns=['customer_id', 'article_id'])\n#         week_df['week'] = i\n    \n#         week_df = week_df.merge(avg_price_week, on=['week'], how='left')\n#         week_df = week_df.merge(\n#             user_week_price.drop(\n#                 columns=['avg_price_1w', 'avg_price_2w', 'avg_price_4w'],\n#                 errors='ignore'\n#             ),\n#             on=['week', 'customer_id'],\n#             how='left'\n#         )\n        \n#         week_df = week_df.merge(\n#             item_week_price.drop(\n#                 columns=['avg_price_1w', 'avg_price_2w', 'avg_price_4w'],\n#                 errors='ignore'\n#             ),\n#             on=['week', 'article_id'],\n#             how='left'\n#         )    \n#         for j in [1, 2, 4]:\n    \n#             last_item_price = (\n#                 item_week_price[item_week_price[f\"diff_item_avg_spend_{j}w\"].notna()]\n#                 .sort_values(\"week\")\n#                 .groupby(\"article_id\")[f\"diff_item_avg_spend_{j}w\"]\n#                 .first()\n#             )\n        \n#             week_df[f\"diff_item_avg_spend_{j}w\"] = week_df[f\"diff_item_avg_spend_{j}w\"].fillna(\n#                 week_df[\"article_id\"].map(last_item_price)\n#             )\n        \n#             week_df[f\"item_avg_price_{j}w\"] = np.where(\n#                 week_df[f\"item_avg_price_{j}w\"].isna(),\n#                 week_df[f\"diff_item_avg_spend_{j}w\"] + week_df[f'avg_price_{j}w'],   # biểu thức trong cùng hàng\n#                 week_df[f\"item_avg_price_{j}w\"]\n#             )\n    \n#             last_user_price = (\n#                 user_week_price[user_week_price[f\"diff_user_avg_spend_{j}w\"].notna()]\n#                 .sort_values(\"week\")\n#                 .groupby(\"customer_id\")[f\"diff_user_avg_spend_{j}w\"]\n#                 .first()\n#             )\n        \n#             week_df[f\"diff_user_avg_spend_{j}w\"] = week_df[f\"diff_user_avg_spend_{j}w\"].fillna(\n#                 week_df[\"customer_id\"].map(last_user_price)\n#             )\n        \n#             week_df[f\"user_avg_spend_{j}w\"] = np.where(\n#                 week_df[f\"user_avg_spend_{j}w\"].isna(),\n#                 week_df[f\"diff_user_avg_spend_{j}w\"] + week_df[f'avg_price_{j}w'],   # biểu thức trong cùng hàng\n#                 week_df[f\"user_avg_spend_{j}w\"]\n#             )\n#             del last_item_price, last_user_price\n#             gc.collect()\n    \n    \n#         item_last_buy_week = (\n#             df_trans[df_trans[\"week\"] > i]\n#             .groupby(\"article_id\", as_index=False)[\"week\"]\n#             .min()\n#             .rename(columns={\"week\": \"item_last_buy_week\"})\n#         )\n#         item_last_buy_week['week_nums_last_item'] = item_last_buy_week['item_last_buy_week'] - i\n    \n#         week_df = week_df.merge(\n#             item_last_buy_week[[\"article_id\", \"week_nums_last_item\"]],\n#             on=\"article_id\",\n#             how=\"left\"\n#         )\n#         del item_last_buy_week, \n#         week_df = week_df.drop(columns=[c for c in week_df.columns if c.startswith(\"diff\")])\n#         week_df = week_df.drop(columns=['week'])\n        \n#         for a in alphas:\n#             tmp = pd.DataFrame(index=week_df.index)\n            \n#             tmp[f\"lower\"] = week_df[\"user_avg_spend_1w\"] - a * std\n#             tmp[f\"upper\"] = week_df[\"user_avg_spend_1w\"] + a * std\n#             mask = False\n\n#             mask |= (week_df[\"item_avg_price_1w\"] >= tmp[f\"lower\"]) & \\\n#                 (week_df[\"item_avg_price_1w\"] <= tmp[f\"upper\"])\n            \n#             candidates = week_df.loc[mask, [\"customer_id\", \"article_id\"]].drop_duplicates()\n#             new_result = evaluate_rule(rule_name=rule_name, \n#                                       candidates=candidates,\n#                                       results=results,\n#                                       valid_exploded=valid_exploded, \n#                                       week=i,\n#                                       alpha=a)\n#             new_result[\"rule\"] = rule_name\n#             new_result[\"alpha\"] = a\n#             new_results.append(new_result)\n#         gc.collect()\n# print(\"save to csv file\")\n# df_new_results = pd.DataFrame(new_results)\n# df_new_results.to_csv(\"item_group_time_history_new_results.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T05:21:04.734305Z","iopub.execute_input":"2026-01-04T05:21:04.734613Z","iopub.status.idle":"2026-01-04T06:07:54.226366Z","shell.execute_reply.started":"2026-01-04T05:21:04.734594Z","shell.execute_reply":"2026-01-04T06:07:54.225369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}