{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13800903,"sourceType":"datasetVersion","datasetId":8622232}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier\nfrom datetime import timedelta\nfrom collections import defaultdict, Counter\nimport numpy as np\nfrom tqdm import tqdm\nimport numpy as np\n\nfrom collections import defaultdict\nimport json\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:55:59.658372Z","iopub.execute_input":"2025-11-27T03:55:59.658650Z","iopub.status.idle":"2025-11-27T03:56:07.579286Z","shell.execute_reply.started":"2025-11-27T03:55:59.658626Z","shell.execute_reply":"2025-11-27T03:56:07.578567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions = pd.read_csv('/kaggle/input/h-and-m-csv-dataset/transactions_train.csv', dtype={'article_id': str}, parse_dates=['t_dat'])\narticles = pd.read_csv('/kaggle/input/h-and-m-csv-dataset/items.csv', dtype={'article_id': str})\ncustomers = pd.read_csv('/kaggle/input/h-and-m-csv-dataset/customers.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:56:07.580277Z","iopub.execute_input":"2025-11-27T03:56:07.580806Z","iopub.status.idle":"2025-11-27T03:58:59.066199Z","shell.execute_reply.started":"2025-11-27T03:56:07.580784Z","shell.execute_reply":"2025-11-27T03:58:59.065595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles['article_id'] = articles['article_id'].astype(str).str.strip().str.zfill(10)\ntransactions['article_id'] = transactions['article_id'].astype(str).str.strip().str.zfill(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:58:59.066886Z","iopub.execute_input":"2025-11-27T03:58:59.067163Z","iopub.status.idle":"2025-11-27T03:59:08.364781Z","shell.execute_reply.started":"2025-11-27T03:58:59.067136Z","shell.execute_reply":"2025-11-27T03:59:08.363971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. Xác định tên cột Gender ---\n# Bạn hãy thay 'gender_column_name' bằng tên thực tế cột giới tính trong file items.csv của bạn\n# Ví dụ: 'perceived_colour_value_id' hoặc tên cột bạn đã tạo sẵn\ngender_col = 'gender'  # <--- THAY TÊN CỘT Ở ĐÂY\n\n# --- 3. Merge & Tính Mean ---\n# Chỉ lấy cột customer_id và article_id từ transactions để tiết kiệm RAM\nmerged_df = transactions[['customer_id', 'article_id']].merge(\n    articles[['article_id', gender_col]], \n    on='article_id', \n    how='left'\n)\n\n# Tính trung bình gender cho mỗi khách hàng\n# Kết quả sẽ là một Series với index là customer_id\ncust_mean_gender = merged_df.groupby('customer_id')[gender_col].mean()\n\n# --- 4. Áp dụng Logic Threshold ---\n# Logic: Mean > 1.5 => 2, ngược lại (<= 1.5) => 1\n# Sử dụng np.where cho tốc độ xử lý nhanh trên dữ liệu lớn\npredicted_values = np.where(cust_mean_gender > 1.5, 2, 1)\n\n# Tạo DataFrame từ kết quả\ncust_prediction = pd.DataFrame({\n    'customer_id': cust_mean_gender.index,\n    'predicted_gender': predicted_values\n})\n\n# --- 5. Merge vào bảng Customer gốc ---\ncustomers = customers.merge(cust_prediction, on='customer_id', how='left')\n\n# --- 6. Xử lý khách hàng mới (Cold Start) ---\n# Những khách hàng chưa từng mua gì sẽ bị NaN. \n# Bạn cần quyết định gán nhãn mặc định (ví dụ: gán theo giới tính phổ biến nhất là 1 hoặc 2, hoặc gán 0 để phân biệt)\n# Ở đây ví dụ gán mặc định là 1\ncustomers['predicted_gender'] = customers['predicted_gender'].fillna(1).astype(int)\n\n# Kiểm tra kết quả\nprint(customers[['customer_id', 'predicted_gender']].head())\nprint(\"Phân phối dự đoán:\\n\", customers['predicted_gender'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:08.366563Z","iopub.execute_input":"2025-11-27T03:59:08.366874Z","iopub.status.idle":"2025-11-27T03:59:29.884556Z","shell.execute_reply.started":"2025-11-27T03:59:08.366856Z","shell.execute_reply":"2025-11-27T03:59:29.883878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Dữ liệu đã tải xong.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:29.885238Z","iopub.execute_input":"2025-11-27T03:59:29.885493Z","iopub.status.idle":"2025-11-27T03:59:29.889947Z","shell.execute_reply.started":"2025-11-27T03:59:29.885476Z","shell.execute_reply":"2025-11-27T03:59:29.889349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Tách dữ liệu theo thời gian ---\n# Tuần cuối cùng dùng để kiểm thử (validation)\nval_start_date = transactions['t_dat'].max() - timedelta(days=6)\n# Tuần trước đó dùng để làm target cho tập huấn luyện (training)\ntrain_target_start_date = val_start_date - timedelta(days=7)\n\n# Dữ liệu lịch sử để tạo ứng viên và đặc trưng\nhistory = transactions[transactions['t_dat'] < train_target_start_date]\n# Dữ liệu target cho tập train\ntrain_targets = transactions[(transactions['t_dat'] >= train_target_start_date) & (transactions['t_dat'] < val_start_date)]\n# Dữ liệu target cho tập validation\nval_targets = transactions[transactions['t_dat'] >= val_start_date]\n\nprint(f\"Lịch sử: {history.shape[0]} giao dịch\")\nprint(f\"Target cho Train: {train_targets.shape[0]} giao dịch\")\nprint(f\"Target cho Validation: {val_targets.shape[0]} giao dịch\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:29.890838Z","iopub.execute_input":"2025-11-27T03:59:29.891624Z","iopub.status.idle":"2025-11-27T03:59:32.155987Z","shell.execute_reply.started":"2025-11-27T03:59:29.891606Z","shell.execute_reply":"2025-11-27T03:59:32.155427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:32.156759Z","iopub.execute_input":"2025-11-27T03:59:32.156997Z","iopub.status.idle":"2025-11-27T03:59:32.178566Z","shell.execute_reply.started":"2025-11-27T03:59:32.156961Z","shell.execute_reply":"2025-11-27T03:59:32.177907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(articles['product_group_name'].unique())\nprint(articles['product_type_name'].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:32.179256Z","iopub.execute_input":"2025-11-27T03:59:32.179442Z","iopub.status.idle":"2025-11-27T03:59:32.194255Z","shell.execute_reply.started":"2025-11-27T03:59:32.179427Z","shell.execute_reply":"2025-11-27T03:59:32.193582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# customers[customers[\"age\"].isna()]\ncust_history = history[history['customer_id'].isin(train_targets['customer_id'])]\n    \ncust_history = pd.merge(cust_history, customers[['customer_id', 'age','predicted_gender']], on='customer_id')\ncust_history = pd.merge(cust_history, articles[['article_id','product_group_name']], on='article_id')\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:32.194937Z","iopub.execute_input":"2025-11-27T03:59:32.195220Z","iopub.status.idle":"2025-11-27T03:59:37.668333Z","shell.execute_reply.started":"2025-11-27T03:59:32.195205Z","shell.execute_reply":"2025-11-27T03:59:37.667731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[['article_id','product_group_name']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:37.670596Z","iopub.execute_input":"2025-11-27T03:59:37.670847Z","iopub.status.idle":"2025-11-27T03:59:37.680457Z","shell.execute_reply.started":"2025-11-27T03:59:37.670832Z","shell.execute_reply":"2025-11-27T03:59:37.679700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cust_history.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:37.681357Z","iopub.execute_input":"2025-11-27T03:59:37.681636Z","iopub.status.idle":"2025-11-27T03:59:37.703762Z","shell.execute_reply.started":"2025-11-27T03:59:37.681611Z","shell.execute_reply":"2025-11-27T03:59:37.702888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_counts = cust_history.isnull().sum()\nprint(nan_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:37.705308Z","iopub.execute_input":"2025-11-27T03:59:37.705875Z","iopub.status.idle":"2025-11-27T03:59:38.455962Z","shell.execute_reply.started":"2025-11-27T03:59:37.705856Z","shell.execute_reply":"2025-11-27T03:59:38.455284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df =cust_history.dropna(subset=['age'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:38.456663Z","iopub.execute_input":"2025-11-27T03:59:38.456878Z","iopub.status.idle":"2025-11-27T03:59:38.835210Z","shell.execute_reply.started":"2025-11-27T03:59:38.456862Z","shell.execute_reply":"2025-11-27T03:59:38.834599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.metrics import silhouette_score\nimport matplotlib.pyplot as plt\n\n\nlabel_encoder = LabelEncoder()\n# df.loc[:, 'product_group_encoded'] = label_encoder.fit_transform(df.loc[:,'product_group_name'])\n# # Chọn features cho clustering\n# features = ['price', 'age', 'product_group_encoded']\n# X = df[features]\n\n# scaler = StandardScaler()\n\n# X_scaled = scaler.fit_transform(X)\n\n# n_clusters=4\n# random_state=42\n# kmeans = KMeans(n_clusters=n_clusters, random_state=random_state, n_init=10)\n# kmeans.fit(X_scaled)\n\n\n\n# wcss = []\n# for i in range(1, 11):\n#     kmeans = KMeans(n_clusters=i, random_state=42, n_init=10)\n#     kmeans.fit(X_scaled)\n#     wcss.append(kmeans.inertia_)\n\n# plt.figure(figsize=(10, 6))\n# plt.plot(range(1, 11), wcss, marker='o')\n# plt.title('Elbow Method for Optimal Number of Clusters')\n# plt.xlabel('Number of Clusters')\n# plt.ylabel('WCSS')\n# plt.show()\n\n# n_clusters=4\n# random_state=42\n# kmeans = KMeans(n_clusters=n_clusters, random_state=random_state)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:38.835913Z","iopub.execute_input":"2025-11-27T03:59:38.836160Z","iopub.status.idle":"2025-11-27T03:59:39.037481Z","shell.execute_reply.started":"2025-11-27T03:59:38.836143Z","shell.execute_reply":"2025-11-27T03:59:39.036942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df.loc[:,'cluster'] = kmeans.labels_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:39.038185Z","iopub.execute_input":"2025-11-27T03:59:39.038434Z","iopub.status.idle":"2025-11-27T03:59:39.041889Z","shell.execute_reply.started":"2025-11-27T03:59:39.038408Z","shell.execute_reply":"2025-11-27T03:59:39.041169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:39.042641Z","iopub.execute_input":"2025-11-27T03:59:39.042908Z","iopub.status.idle":"2025-11-27T03:59:39.057073Z","shell.execute_reply.started":"2025-11-27T03:59:39.042892Z","shell.execute_reply":"2025-11-27T03:59:39.056564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# N_CANDIDATES = 100 \n# n_clusters đã được định nghĩa là 4 trong code của bạn\n# n_per_cluster = N_CANDIDATES // n_clusters # Dùng // để chia lấy phần nguyên","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:39.057793Z","iopub.execute_input":"2025-11-27T03:59:39.057986Z","iopub.status.idle":"2025-11-27T03:59:39.070008Z","shell.execute_reply.started":"2025-11-27T03:59:39.057972Z","shell.execute_reply":"2025-11-27T03:59:39.069551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(n_per_cluster)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:39.071394Z","iopub.execute_input":"2025-11-27T03:59:39.071556Z","iopub.status.idle":"2025-11-27T03:59:39.085897Z","shell.execute_reply.started":"2025-11-27T03:59:39.071544Z","shell.execute_reply":"2025-11-27T03:59:39.085250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Đếm số lần mua 'article_id' trong TỪNG CỤM\n# Kết quả là một Series với MultiIndex (cluster, article_id)\n# cluster_article_counts = df.groupby('cluster')['article_id'].value_counts().rename('purchase_count')\n\n# # 4. Với mỗi cụm (level='cluster'), lấy ra top 'n_per_cluster' article_id\n# # có số lần mua cao nhất\n# top_articles_series = cluster_article_counts.groupby(level='cluster').nlargest(n_per_cluster)\n\n# top_articles_series.index = top_articles_series.index.droplevel(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:39.086577Z","iopub.execute_input":"2025-11-27T03:59:39.086747Z","iopub.status.idle":"2025-11-27T03:59:39.101296Z","shell.execute_reply.started":"2025-11-27T03:59:39.086733Z","shell.execute_reply":"2025-11-27T03:59:39.100642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(top_articles_series.index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:39.101973Z","iopub.execute_input":"2025-11-27T03:59:39.102273Z","iopub.status.idle":"2025-11-27T03:59:39.115513Z","shell.execute_reply.started":"2025-11-27T03:59:39.102257Z","shell.execute_reply":"2025-11-27T03:59:39.114993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# top_articles_df = top_articles_series.reset_index()\n\n# Lấy danh sách các article_id duy nhất từ DataFrame kết quả\n# final_candidate_list = list(top_articles_df['article_id'].unique())\n\n# In kết quả\n# print(f\"Số lượng bài báo đề xuất mỗi cụm: {n_per_cluster}\")\n# print(f\"Tổng số bài báo ứng viên duy nhất: {len(final_candidate_list)}\")\n# print(\"Danh sách ID bài báo ứng viên:\")\n# print(final_candidate_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:59:39.116182Z","iopub.execute_input":"2025-11-27T03:59:39.116369Z","iopub.status.idle":"2025-11-27T03:59:39.130561Z","shell.execute_reply.started":"2025-11-27T03:59:39.116356Z","shell.execute_reply":"2025-11-27T03:59:39.130025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from abc import ABC, abstractmethod\nfrom typing import Dict, List, Set\nimport pandas as pd\nfrom datetime import timedelta\nfrom collections import defaultdict, Counter\nimport numpy as np\nfrom tqdm import tqdm\nimport cudf\nimport cupy as cp\nimport cuml\nfrom cuml.neighbors import NearestNeighbors\nfrom cupyx.scipy.sparse import csr_matrix as cp_csr_matrix\n\n# Assuming N_CANDIDATES is defined elsewhere or will be passed as a parameter.\n# Using the value from the notebook state for consistency.\nN_CANDIDATES = 100\n\n\nclass GlobalRetrieveRule(ABC):\n    \"\"\"Use certain rules to retrieve items for all customers.\"\"\"\n\n    def merge(self, result: pd.DataFrame):\n        result = result[[self.iid, \"method\", \"score\"]]\n\n        num_item = result.shape[0]\n        num_user = self.customer_list.shape[0]\n\n        tmp_user = np.repeat(self.customer_list, num_item)\n        tmp_df = result.iloc[np.tile(np.arange(num_item), num_user)]\n        tmp_df = tmp_df.reset_index(drop=True)\n        tmp_df[\"customer_id\"] = tmp_user\n\n        return tmp_df\n\n    @abstractmethod\n    def retrieve(self) -> pd.DataFrame:\n        \"\"\"Retrieve items\n\n        Returns:\n            pd.DataFrame: (article_id, method, score)\n        \"\"\"\n        pass\n\n\nclass FilterRule(ABC):\n    \"\"\"Use certain rules to remove some retrieved items.\"\"\"\n\n    @abstractmethod\n    def retrieve(self) -> List:\n        \"\"\"Retrieve items\n\n        Returns:\n            List: items to be removed\n        \"\"\"\n        pass\n\n\nclass PersonalRetrieveRule(ABC):\n    \"\"\"Use certain rules to retrieve items for individual customers.\"\"\"\n\n    @abstractmethod\n    def retrieve(self, customer_list: List[str]) -> Dict[str, List[str]]:\n        \"\"\"Retrieve items for specific customers.\n\n        Parameters\n        ----------\n        customer_list : List[str]\n            A list of customer IDs for whom to retrieve items.\n\n        Returns:\n            Dict[str, List[str]]: A dictionary where keys are customer_id and values are\n                                  lists of article_ids (candidates).\n        \"\"\"\n        pass\n\n# * ======================= Personal Retrieve Rules ======================= *\n\n\nclass OrderHistory(PersonalRetrieveRule):\n    \"\"\"Retrieve recently bought items by the customer.\"\"\"\n\n    def __init__(\n        self,\n        self_trans_df: pd.DataFrame,\n        days: int = 7,\n        n: int = None,\n        name: str = \"1\",\n        item_id: str = \"article_id\",\n    ):\n        \"\"\"Initialize OrderHistory.\"\"\"\n        self.iid = item_id\n        self.trans_df = self_trans_df[[\"t_dat\", \"customer_id\", item_id]]\n        self.days = days\n\nclass UserToUserCandidateGenerator(PersonalRetrieveRule):\n    \"\"\"Suggests articles from similar users based on shared product groups (User-to-User collaborative filtering).\"\"\"\n\n    def __init__(self, history_df: pd.DataFrame, articles_df: pd.DataFrame, \n                 days_window: int = 30, k_neighbors: int = 3, n_candidates: int = 100):\n        # Chuyển input từ pandas sang cudf ngay khi khởi tạo để xử lý trên GPU\n        self.history_df = cudf.from_pandas(history_df) if isinstance(history_df, pd.DataFrame) else history_df\n        self.articles_df = cudf.from_pandas(articles_df) if isinstance(articles_df, pd.DataFrame) else articles_df\n        \n        self.days_window = days_window\n        self.k_neighbors = k_neighbors\n        self.n_candidates = n_candidates\n        self.iid = 'article_id'\n        # [MODIFY] Thêm trường method vào output columns mặc định\n        self.output_columns = ['customer_id', 'article_id', 'method', 'score']\n\n    def retrieve(self, customer_list: List[str]) -> pd.DataFrame:\n        # 1. Lọc dữ liệu theo thời gian (trên GPU)\n        last_date = self.history_df['t_dat'].max()\n        start_date_window = last_date - pd.Timedelta(days=self.days_window)\n        recent_df = self.history_df[self.history_df['t_dat'] >= start_date_window].copy()\n\n        # 2. Merge để lấy product_group_name\n        articles_subset = self.articles_df[[self.iid, 'product_group_name']]\n        recent_with_groups = recent_df.merge(articles_subset, on=self.iid, how='inner')\n\n        if len(recent_with_groups) == 0:\n            return pd.DataFrame(columns=self.output_columns)\n\n        # 3. Tạo Sparse Matrix (User x ProductGroup)\n        recent_with_groups['user_code'], user_index = recent_with_groups['customer_id'].factorize()\n        recent_with_groups['group_code'], group_index = recent_with_groups['product_group_name'].factorize()\n        \n        unique_interactions = recent_with_groups[['user_code', 'group_code']].drop_duplicates()\n        \n        rows = unique_interactions['user_code'].values\n        cols = unique_interactions['group_code'].values\n        data = cp.ones(len(rows), dtype=cp.float32)\n        \n        n_users = len(user_index)\n        n_groups = len(group_index)\n        \n        user_group_matrix = cp_csr_matrix((data, (rows, cols)), shape=(n_users, n_groups))\n\n        # 4. Tìm K-Nearest Neighbors bằng cuML\n        n_neighbors_actual = min(self.k_neighbors + 1, n_users)\n        if n_neighbors_actual <= 1:\n             return pd.DataFrame(columns=self.output_columns)\n\n        knn_model = NearestNeighbors(n_neighbors=n_neighbors_actual, metric='cosine', algorithm='brute')\n        knn_model.fit(user_group_matrix)\n\n        # Lọc ra danh sách user cần recommend\n        target_users_cudf = cudf.Series(customer_list)\n        user_map = cudf.DataFrame({'customer_id': user_index, 'code': cp.arange(len(user_index))})\n        target_indices = target_users_cudf.to_frame('customer_id').merge(user_map, on='customer_id', how='inner')['code']\n        \n        if len(target_indices) == 0:\n            return pd.DataFrame(columns=self.output_columns)\n\n        target_matrix = user_group_matrix[target_indices.values.get(), :]\n        \n        distances, indices = knn_model.kneighbors(target_matrix, n_neighbors=n_neighbors_actual)\n        \n        # 5. Candidate Generation\n        neighbor_indices = indices[:, 1:] # Bỏ cột 0 (chính user đó)\n        \n        n_targets = len(target_indices)\n        k_real = neighbor_indices.shape[1]\n        \n        neighbors_flat = neighbor_indices.flatten()\n        targets_repeated = cp.repeat(target_indices.values, k_real)\n        \n        df_neighbors = cudf.DataFrame({\n            'user_code': targets_repeated,\n            'neighbor_code': neighbors_flat\n        })\n        \n        history_subset = recent_with_groups[['user_code', self.iid]].drop_duplicates()\n        \n        candidates = df_neighbors.merge(\n            history_subset, \n            left_on='neighbor_code', \n            right_on='user_code', \n            how='inner'\n        ).rename(columns={self.iid: 'article_id', 'user_code_x': 'user_code'})\n        \n        # 6. Anti-join\n        target_history = recent_with_groups[['user_code', self.iid]].drop_duplicates()\n        candidates = candidates.merge(\n            target_history,\n            on=['user_code', 'article_id'],\n            how='leftanti'\n        )\n        \n        # 7. Tính điểm (Score)\n        scored_candidates = candidates.groupby(['user_code', 'article_id']).size().reset_index(name='count')\n        scored_candidates['score'] = scored_candidates['count'] / self.k_neighbors\n        \n        # 8. Lấy Top N candidates\n        scored_candidates = scored_candidates.sort_values(['user_code', 'score'], ascending=[True, False])\n        result_gpu = scored_candidates.groupby('user_code').head(self.n_candidates)\n        \n        # 9. Chuyển đổi mã user_code\n        final_result = result_gpu.merge(user_map, left_on='user_code', right_on='code', how='left')\n        \n        # [MODIFY] Thêm cột method và sắp xếp lại output\n        final_result['method'] = 'u2u'\n        \n        # Chọn các cột theo đúng yêu cầu\n        final_result = final_result[['customer_id', 'article_id', 'method', 'score']]\n        \n        return final_result.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T04:20:56.639496Z","iopub.execute_input":"2025-11-27T04:20:56.640152Z","iopub.status.idle":"2025-11-27T04:20:56.661212Z","shell.execute_reply.started":"2025-11-27T04:20:56.640130Z","shell.execute_reply":"2025-11-27T04:20:56.660521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = UserToUserCandidateGenerator(history, articles,days_window =30, k_neighbors = 350)\ncandidate = model.retrieve(train_targets['customer_id'].unique().tolist())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T04:20:56.869080Z","iopub.execute_input":"2025-11-27T04:20:56.869733Z","iopub.status.idle":"2025-11-27T04:21:13.370676Z","shell.execute_reply.started":"2025-11-27T04:20:56.869712Z","shell.execute_reply":"2025-11-27T04:21:13.370084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"candidate.to_csv('u2u_candidates.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T04:21:13.371686Z","iopub.execute_input":"2025-11-27T04:21:13.371906Z","iopub.status.idle":"2025-11-27T04:21:26.937031Z","shell.execute_reply.started":"2025-11-27T04:21:13.371881Z","shell.execute_reply":"2025-11-27T04:21:26.936047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AgeGroupBestSellersRecall(PersonalRetrieveRule):\n    \"\"\"Retrieve top N best-selling items by age group that customer hasn't purchased.\"\"\"\n    \n    def __init__(\n        self,\n        trans_df: pd.DataFrame,\n        customers_df: pd.DataFrame,\n        n_candidates: int = N_CANDIDATES,\n        age_groups: List[tuple] = None,\n        name: str = \"age_group_best_sellers\",\n        item_id: str = \"article_id\",\n    ):\n        \"\"\"Initialize AgeGroupBestSellersRecall.\n        \n        Parameters\n        ----------\n        trans_df : pd.DataFrame\n            Dataframe of transaction records.\n        customers_df : pd.DataFrame\n            Dataframe containing customer information with age.\n        n_candidates : int, optional\n            Number of candidate items to retrieve, by default N_CANDIDATES\n        age_groups : List[tuple], optional\n            List of age group tuples, e.g., [(21, 30), (31, 40), ...]\n            If None, uses default age groups.\n        name : str, optional\n            Name of the rule, by default \"age_group_best_sellers\"\n        item_id : str, optional\n            Name of item id, by default \"article_id\"\n        \"\"\"\n        self.iid = item_id\n        self.trans_df = trans_df[[\"t_dat\", \"customer_id\", item_id]].copy()\n        self.customers_df = customers_df[[\"customer_id\", \"age\"]].copy()\n        self.n_candidates = n_candidates\n        self.name = name\n        \n        # Define default age groups if not provided\n        if age_groups is None:\n            self.age_groups = [\n                (16, 20), (21, 25), (26, 30), (31, 35), \n                (36, 40), (41, 45), (46, 50), (51, 55), \n                (56, 60), (61, 65), (66, 70), (71, 75), (76, 99)\n            ]\n        else:\n            self.age_groups = age_groups\n        \n        # Precompute best sellers by age group\n        self.age_group_best_sellers = self._compute_age_group_best_sellers()\n        self.customer_purchased_items = self._get_customer_purchased_items()\n    \n    def _get_age_group(self, age: float) -> str:\n        \"\"\"Map age to age group string.\"\"\"\n        if pd.isna(age):\n            return \"unknown\"\n        \n        for min_age, max_age in self.age_groups:\n            if min_age <= age <= max_age:\n                return f\"{min_age}-{max_age}\"\n        return \"unknown\"\n    \n    def _compute_age_group_best_sellers(self) -> Dict[str, List[str]]:\n        \"\"\"Compute best-selling items for each age group.\"\"\"\n        print(\"Computing best sellers by age group...\")\n        \n        # Merge transactions with customer age information\n        merged_df = self.trans_df.merge(\n            self.customers_df[[\"customer_id\", \"age\"]], \n            on=\"customer_id\", \n            how=\"inner\"\n        )\n        \n        # Add age group column\n        merged_df[\"age_group\"] = merged_df[\"age\"].apply(self._get_age_group)\n        \n        # Calculate best sellers for each age group\n        age_group_best_sellers = {}\n        \n        for age_group in tqdm(merged_df[\"age_group\"].unique(), desc=\"Processing age groups\"):\n            if age_group == \"unknown\":\n                continue\n                \n            age_group_trans = merged_df[merged_df[\"age_group\"] == age_group]\n            \n            # Count item frequencies and get top N\n            item_counts = age_group_trans[self.iid].value_counts().head(self.n_candidates)\n            best_sellers = item_counts.index.tolist()\n            \n            age_group_best_sellers[age_group] = best_sellers\n        \n        # Handle unknown age group separately\n        unknown_trans = merged_df[merged_df[\"age_group\"] == \"unknown\"]\n        if not unknown_trans.empty:\n            item_counts = unknown_trans[self.iid].value_counts().head(self.n_candidates)\n            age_group_best_sellers[\"unknown\"] = item_counts.index.tolist()\n        \n        return age_group_best_sellers\n    \n    def _get_customer_purchased_items(self) -> Dict[str, Set[str]]:\n        \"\"\"Get purchased items for each customer.\"\"\"\n        print(\"Computing customer purchased items...\")\n        customer_purchased = {}\n        \n        for customer_id, group in tqdm(self.trans_df.groupby(\"customer_id\"), desc=\"Processing customers\"):\n            customer_purchased[customer_id] = set(group[self.iid].unique())\n        \n        return customer_purchased\n    \n    def _get_customer_age_group(self, customer_id: str) -> str:\n        \"\"\"Get age group for a specific customer.\"\"\"\n        customer_info = self.customers_df[self.customers_df[\"customer_id\"] == customer_id]\n        \n        if customer_info.empty:\n            return \"unknown\"\n        \n        age = customer_info[\"age\"].iloc[0]\n        return self._get_age_group(age)\n    \n    def retrieve(self, customer_list: List[str]) -> Dict[str, List[str]]:\n        \"\"\"Retrieve best-selling items by age group that customer hasn't purchased.\n        \n        Parameters\n        ----------\n        customer_list : List[str]\n            A list of customer IDs for whom to retrieve items.\n            \n        Returns\n        -------\n        Dict[str, List[str]]\n            Dictionary mapping customer_id to list of candidate article_ids\n        \"\"\"\n        print(f\"Retrieving age-group based best sellers for {len(customer_list)} customers...\")\n        \n        results = {}\n        \n        for customer_id in tqdm(customer_list, desc=\"Retrieving candidates\"):\n            # Get customer's age group\n            age_group = self._get_customer_age_group(customer_id)\n            \n            # Get best sellers for the age group\n            if age_group in self.age_group_best_sellers:\n                age_group_items = self.age_group_best_sellers[age_group]\n            else:\n                # Fallback to overall best sellers if age group not found\n                age_group_items = self.age_group_best_sellers.get(\n                    list(self.age_group_best_sellers.keys())[0], []\n                )\n            \n            # Get customer's purchased items\n            purchased_items = self.customer_purchased_items.get(customer_id, set())\n            \n            # Filter out items customer has already purchased\n            candidate_items = [\n                item for item in age_group_items \n                if item not in purchased_items\n            ][:self.n_candidates]\n            \n            results[customer_id] = candidate_items\n        \n        return results\n\n# Example usage:\nif __name__ == \"__main__\":\n    # Example data\n    trans_data = {\n        't_dat': ['2020-01-01', '2020-01-02', '2020-01-03'] * 100,\n        'customer_id': [f'cust_{i}' for i in range(1, 101)] * 3,\n        'article_id': [f'item_{i}' for i in range(1, 301)]\n    }\n    \n    customer_data = {\n        'customer_id': [f'cust_{i}' for i in range(1, 101)],\n        'age': np.random.randint(18, 70, 100)\n    }\n    \n    trans_df = pd.DataFrame(trans_data)\n    customers_df = pd.DataFrame(customer_data)\n    \n    # Define custom age groups\n    custom_age_groups = [(18, 25), (26, 35), (36, 45), (46, 55), (56, 70)]\n    \n    # Initialize the recall class\n    age_group_recall = AgeGroupBestSellersRecall(\n        trans_df=trans_df,\n        customers_df=customers_df,\n        n_candidates=50,\n        age_groups=custom_age_groups\n    )\n    \n    # Retrieve candidates for a list of customers\n    customer_list = [f'cust_{i}' for i in range(1, 11)]\n    candidates = age_group_recall.retrieve(customer_list)\n    \n    print(f\"Retrieved candidates for {len(candidates)} customers\")\n    for cust_id, items in list(candidates.items())[:3]:\n        print(f\"Customer {cust_id}: {len(items)} candidates\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Thêm dòng này để lưu file ---\n# candidate.to_csv('u2u_candidates.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:00:55.185838Z","iopub.execute_input":"2025-11-26T07:00:55.186074Z","iopub.status.idle":"2025-11-26T07:00:55.200610Z","shell.execute_reply.started":"2025-11-26T07:00:55.186053Z","shell.execute_reply":"2025-11-26T07:00:55.199853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndef evaluate_recall(candidates_df, test_df, user_col='customer_id', item_col='article_id', pred_col='article_id'):\n    \"\"\"\n    Hàm tính % test có trong candidate (Recall).\n    \n    Args:\n        candidates_df: DataFrame chứa customer_id và list các article_id gợi ý.\n        test_df: DataFrame chứa customer_id và article_id thực tế (ground truth).\n        user_col: Tên cột user (mặc định: customer_id).\n        item_col: Tên cột item trong tập Test (mặc định: article_id).\n        pred_col: Tên cột chứa list gợi ý trong tập Candidate (mặc định: article_id).\n        \n    Returns:\n        recall: Tỷ lệ phần trăm (%) item test nằm trong danh sách gợi ý.\n        merged_df: DataFrame chi tiết kết quả (để debug nếu cần).\n    \"\"\"\n    print(candidates_df.columns)\n    print(test_df.columns)\n    \n    # 1. Đổi tên cột gợi ý trong candidates để tránh trùng lặp khi merge\n    # Giả sử candidates_df có cột 'article_id' là list các gợi ý\n    candidates_clean = candidates_df[[user_col, pred_col]].copy()\n    candidates_clean = candidates_clean.rename(columns={pred_col: 'recommended_items'})\n    \n    # 2. Merge tập Test với tập Candidate (Left Join)\n    # Dùng Left Join để giữ lại tất cả các dòng trong Test\n    merged_df = pd.merge(test_df, candidates_clean, on=user_col, how='left')\n    \n    # 3. Hàm kiểm tra xem item thật (test) có nằm trong list gợi ý không\n    def check_hit(row):\n        true_item = row[item_col]\n        recommendations = row['recommended_items']\n        \n        # Nếu không có gợi ý nào (NaN) hoặc list rỗng -> False\n        if not isinstance(recommendations, list) and pd.isna(recommendations):\n            return 0\n        \n        # Kiểm tra sự tồn tại\n        return 1 if true_item in recommendations else 0\n\n    # Áp dụng hàm check\n    merged_df['is_hit'] = merged_df.apply(check_hit, axis=1)\n    \n    # 4. Tính toán chỉ số\n    total_hits = merged_df['is_hit'].sum()\n    total_test_samples = len(test_df)\n    \n    recall = total_hits / total_test_samples\n    \n    print(f\"--- Kết quả Đánh giá ---\")\n    print(f\"Tổng số mẫu trong Test: {total_test_samples}\")\n    print(f\"Số mẫu Test được tìm thấy trong Candidate: {total_hits}\")\n    print(f\"Tỷ lệ (Recall): {recall:.4f} ({recall*100:.2f}%)\")\n    \n    return recall, merged_df\n\n# # --- Ví dụ cách sử dụng ---\n\n# # Giả sử tập candidate của bạn (lưu ý cột article_id đang là list)\n# # candidate = ... (Dữ liệu từ model.retrieve của bạn)\n\n# # Giả sử tập test của bạn\n# # test = ... \n\n# # Gọi hàm\n# score, details = evaluate_recall(candidate, train_targets)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:00:55.201254Z","iopub.execute_input":"2025-11-26T07:00:55.201440Z","iopub.status.idle":"2025-11-26T07:00:55.214836Z","shell.execute_reply.started":"2025-11-26T07:00:55.201427Z","shell.execute_reply":"2025-11-26T07:00:55.214240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\ndef run_random_search(\n    train_df, \n    articles_df, \n    test_df, \n    customer_list, \n    n_iter=10,            # Số lần thử nghiệm\n    k_range=(5, 200),     # Khoảng tìm kiếm số neighbor\n    n_candidates=100      # Số lượng candidate cần lấy\n):\n    \"\"\"\n    Thực hiện Random Search để tìm k_neighbors tốt nhất.\n    \"\"\"\n    \n    best_score = -1\n    best_k = -1\n    results_log = []\n    \n    print(f\"--- Bắt đầu Random Search ({n_iter} lượt) ---\")\n    \n    for i in range(n_iter):\n        # 1. Random chọn k trong khoảng range\n        current_k = random.randint(k_range[0], k_range[1])\n        \n        print(f\"\\n[Iter {i+1}/{n_iter}] Testing k_neighbors = {current_k}...\")\n        \n        # 2. Khởi tạo model với k hiện tại\n        # Lưu ý: Dùng class GPU (UserToUserCandidateGeneratorGPU) để chạy cho nhanh\n        # Nếu dùng CPU thì thay tên class tương ứng\n        model = UserToUserCandidateGenerator(\n            history_df=train_df,\n            articles_df=articles_df,\n            days_window=30,\n            k_neighbors=current_k,  # <--- Tham số đang tune\n            n_candidates=n_candidates\n        )\n        \n        # 3. Retrieve candidates\n        # Kết quả trả về dạng: customer_id | article_id | score\n        candidates_long = model.retrieve(customer_list)\n        \n        if candidates_long.empty:\n            print(f\"Warning: k={current_k} không tìm thấy candidate nào.\")\n            continue\n\n        # 4. CHUYỂN ĐỔI FORMAT: Từ Long sang List để khớp với hàm evaluate_recall\n        # Gom các article_id thành list cho mỗi user\n        candidates_grouped = candidates_long.groupby('customer_id')['article_id'].apply(list).reset_index()\n        \n        # 5. Đánh giá bằng hàm evaluate_recall của bạn\n        recall_score, _ = evaluate_recall(\n            candidates_df=candidates_grouped, \n            test_df=test_df,\n            user_col='customer_id',\n            item_col='article_id', # Tên cột item thật trong test_df\n            pred_col='article_id'  # Tên cột chứa list gợi ý trong candidates_grouped\n        )\n        \n        # 6. Log kết quả\n        results_log.append({'k': current_k, 'recall': recall_score})\n        \n        # 7. Cập nhật Best Score\n        if recall_score > best_score:\n            best_score = recall_score\n            best_k = current_k\n            print(f\"--> Kỷ lục mới! k={best_k} với Recall={best_score:.4f}\")\n            \n    print(\"\\n================ KẾT THÚC ================\")\n    print(f\"Best k_neighbors: {best_k}\")\n    print(f\"Best Recall: {best_score:.4f}\")\n    \n    return best_k, pd.DataFrame(results_log)\n\n# --- HƯỚNG DẪN CHẠY ---\n\n# 1. Chuẩn bị dữ liệu\n# unique_customers_test = test_df['customer_id'].unique().tolist()\n\n# 2. Chạy search\n# best_k, log_df = run_random_search(\n#     train_df=train_df,           # Lịch sử mua hàng\n#     articles_df=articles_df,     # Thông tin sản phẩm\n#     test_df=test_df,             # Tập ground truth để chấm điểm\n#     customer_list=unique_customers_test, # Danh sách user cần dự đoán\n#     n_iter=5,                    # Thử 5 giá trị ngẫu nhiên\n#     k_range=(10, 100)            # Random k từ 10 đến 100\n# )\n\n# 3. Xem biểu đồ kết quả (nếu muốn)\n# log_df.sort_values('recall', ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:00:55.215506Z","iopub.execute_input":"2025-11-26T07:00:55.215727Z","iopub.status.idle":"2025-11-26T07:00:55.232744Z","shell.execute_reply.started":"2025-11-26T07:00:55.215707Z","shell.execute_reply":"2025-11-26T07:00:55.232090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- HƯỚNG DẪN CHẠY ---\n\n# 1. Chuẩn bị dữ liệu\nunique_customers_test = train_targets['customer_id'].unique().tolist()\n\n# 2. Chạy search\nbest_k, log_df = run_random_search(\n    train_df=history,           # Lịch sử mua hàng\n    articles_df=articles,     # Thông tin sản phẩm\n    test_df=train_targets,             # Tập ground truth để chấm điểm\n    customer_list=unique_customers_test, # Danh sách user cần dự đoán\n    n_iter=20,                    # Thử 5 giá trị ngẫu nhiên\n    k_range=(10, 400)            # Random k từ 10 đến 100\n)\n\n# 3. Xem biểu đồ kết quả (nếu muốn)\nlog_df.sort_values('recall', ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:00:55.233346Z","iopub.execute_input":"2025-11-26T07:00:55.233529Z","iopub.status.idle":"2025-11-26T07:06:54.331120Z","shell.execute_reply.started":"2025-11-26T07:00:55.233515Z","shell.execute_reply":"2025-11-26T07:06:54.330533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================================================================\n# # 3. CÁC HÀM TẠO ĐẶC TRƯNG (FEATURE ENGINEERING)\n# # =============================================================================\n# def create_features(df, history_df, articles_df, customers_df):\n#     \"\"\"Hàm tổng hợp để thêm các đặc trưng vào DataFrame ứng viên.\"\"\"\n#     # --- Đặc trưng về khách hàng ---\n#     df = pd.merge(df, customers_df[['customer_id', 'age']], on='customer_id', how='left')\n#     user_activity = history_df.groupby('customer_id')['article_id'].count().reset_index()\n#     user_activity.rename(columns={'article_id': 'user_purchase_count'}, inplace=True)\n#     df = pd.merge(df, user_activity, on='customer_id', how='left')\n\n#     # --- Đặc trưng về sản phẩm ---\n#     item_popularity = history_df.groupby('article_id')['customer_id'].count().reset_index()\n#     item_popularity.rename(columns={'customer_id': 'item_popularity'}, inplace=True)\n#     df = pd.merge(df, item_popularity, on='article_id', how='left')\n    \n#     # --- Đặc trưng tương tác User-Item ---\n#     # Ví dụ: Số lần user đã mua item này\n#     user_item_counts = history_df.groupby(['customer_id', 'article_id']).size().reset_index(name='user_item_purchase_count')\n#     df = pd.merge(df, user_item_counts, on=['customer_id', 'article_id'], how='left')\n    \n#     df.fillna(0, inplace=True)\n#     return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.334591Z","iopub.execute_input":"2025-11-26T07:06:54.334998Z","iopub.status.idle":"2025-11-26T07:06:54.338488Z","shell.execute_reply.started":"2025-11-26T07:06:54.334981Z","shell.execute_reply":"2025-11-26T07:06:54.337873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================================================================\n# # 4. PIPELINE TẠO DỮ LIỆU\n# # =============================================================================\n# def create_dataset(target_df, history_df, articles_df, customers_df):\n#     \"\"\"\n#     Quy trình hoàn chỉnh để tạo ra tập dữ liệu có nhãn và đặc trưng.\n#     \"\"\"\n#     target_customers = target_df['customer_id'].unique()\n    \n#     # --- Tạo ứng viên ---\n#     popular_items = generate_popular_items(history_df)\n#     print(popular_items)\n#     repurchase_cands = generate_repurchase_candidates(history_df, target_customers)\n#     same_prod_cands = generate_same_product_code_candidates(history_df, target_customers, articles_df)\n    \n#     candidate_data = []\n#     for cust_id in target_customers:\n#         # Kết hợp các nguồn ứng viên\n#         candidates = set(popular_items)\n#         candidates.update(repurchase_cands.get(cust_id, []))\n#         candidates.update(same_prod_cands.get(cust_id, []))\n        \n#         for art_id in candidates:\n#             candidate_data.append({'customer_id': cust_id, 'article_id': art_id})\n    \n#     df = pd.DataFrame(candidate_data)\n    \n#     # --- Tạo nhãn ---\n#     purchases = target_df.groupby('customer_id')['article_id'].apply(list).reset_index()\n#     purchases.rename(columns={'article_id': 'purchased'}, inplace=True)\n    \n#     df = pd.merge(df, purchases, on='customer_id', how='left')\n#     df['purchased'] = df['purchased'].fillna(lambda: [])\n#     df['label'] = df.apply(lambda row: 1 if row['article_id'] in row['purchased'] else 0, axis=1)\n    \n#     # --- Tạo đặc trưng ---\n#     df = create_features(df, history_df, articles_df, customers_df)\n    \n#     # Dọn dẹp\n#     df.drop(columns=['purchased'], inplace=True)\n    \n#     return df\n\n# print(\"\\nBắt đầu pipeline tạo dữ liệu...\")\n# # Tạo tập validation\n# val_df = create_dataset(val_targets, transactions[transactions['t_dat'] < val_start_date], articles, customers)\n# print(f\"Đã tạo tập validation với {val_df.shape[0]} dòng và {val_df['label'].sum()} mẫu dương.\")\n# # Tạo tập train\n# train_df = create_dataset(train_targets, history, articles, customers)\n# print(f\"Đã tạo tập train với {train_df.shape[0]} dòng và {train_df['label'].sum()} mẫu dương.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.339153Z","iopub.execute_input":"2025-11-26T07:06:54.339376Z","iopub.status.idle":"2025-11-26T07:06:54.357543Z","shell.execute_reply.started":"2025-11-26T07:06:54.339359Z","shell.execute_reply":"2025-11-26T07:06:54.356943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================================================================\n# # 5. HUẤN LUYỆN MÔ HÌNH VÀ ENSEMBLE\n# # =============================================================================\n# print(\"\\nBắt đầu huấn luyện...\")\n\n# # --- Giảm mẫu âm cho tập train ---\n# pos_samples = train_df[train_df['label'] == 1]\n# neg_samples = train_df[train_df['label'] == 0].sample(n=len(pos_samples) * 5, random_state=42)\n# train_downsampled = pd.concat([pos_samples, neg_samples])\n\n# features = ['age', 'user_purchase_count', 'item_popularity', 'user_item_purchase_count']\n# X_train = train_downsampled[features]\n# y_train = train_downsampled['label']\n\n# # --- Model 1: LightGBM ---\n# print(\"Đang huấn luyện LightGBM...\")\n# lgb_model = lgb.LGBMClassifier(objective='binary', random_state=42)\n# lgb_model.fit(X_train, y_train)\n\n# # --- Model 2: CatBoost ---\n# print(\"Đang huấn luyện CatBoost...\")\n# cat_model = CatBoostClassifier(iterations=200, verbose=0, random_state=42)\n# cat_model.fit(X_train, y_train)\n\n# # --- Dự đoán và Ensemble ---\n# X_val = val_df[features]\n# pred_lgb = lgb_model.predict_proba(X_val)[:, 1]\n# pred_cat = cat_model.predict_proba(X_val)[:, 1]\n\n# # Ensemble bằng cách lấy trung bình\n# val_df['prediction'] = (5* pred_lgb + 7*pred_cat) / 12\n# print(\"Huấn luyện và dự đoán hoàn tất.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.358350Z","iopub.execute_input":"2025-11-26T07:06:54.358565Z","iopub.status.idle":"2025-11-26T07:06:54.379342Z","shell.execute_reply.started":"2025-11-26T07:06:54.358543Z","shell.execute_reply":"2025-11-26T07:06:54.378746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================================================================\n# # 6. ĐÁNH GIÁ\n# # =============================================================================\n# print(\"\\nBắt đầu đánh giá...\")\n# val_df_sorted = val_df.sort_values(['customer_id', 'prediction'], ascending=[True, False])\n\n# top12_recs = val_df_sorted.groupby('customer_id')['article_id'].apply(list).apply(lambda x: x[:12]).reset_index()\n# top12_recs.rename(columns={'article_id': 'recommendations'}, inplace=True)\n\n# actual_purchases = val_targets.groupby('customer_id')['article_id'].apply(list).reset_index()\n# actual_purchases.rename(columns={'article_id': 'purchased_articles'}, inplace=True)\n\n# result_df = pd.merge(top12_recs, actual_purchases, on='customer_id')\n\n# def map_at_12(row):\n#     relevant = row['purchased_articles']\n#     recommended = row['recommendations']\n#     if not relevant: return 0.0\n    \n#     score, num_hits = 0.0, 0.0\n#     for i, p in enumerate(recommended):\n#         if p in relevant and p not in recommended[:i]:\n#             num_hits += 1.0\n#             score += num_hits / (i + 1.0)\n#     return score / min(len(relevant), 12)\n\n# mean_map_score = result_df.apply(map_at_12, axis=1).mean()\n\n# print(\"\\n--- KẾT QUẢ ---\")\n# print(f\"Điểm MAP@12 trên tập validation (sau khi ensemble): {mean_map_score:.6f}\")\n# print(\"\\nVí dụ về gợi ý:\")\n# print(result_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.380020Z","iopub.execute_input":"2025-11-26T07:06:54.380297Z","iopub.status.idle":"2025-11-26T07:06:54.396737Z","shell.execute_reply.started":"2025-11-26T07:06:54.380278Z","shell.execute_reply":"2025-11-26T07:06:54.396074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# res_val_df = pd.concat([val_df, train_df], axis=0)\n# res_val = res_val_df[features]\n\n# pred_lgb = lgb_model.predict_proba(res_val)[:, 1]\n# pred_cat = cat_model.predict_proba(res_val)[:, 1]\n\n\n# # Ensemble bằng cách lấy trung bình\n# res_val_df['prediction'] = (5* pred_lgb + 7*pred_cat) / 12\n# print(\"Huấn luyện và dự đoán hoàn tất.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.397367Z","iopub.execute_input":"2025-11-26T07:06:54.397571Z","iopub.status.idle":"2025-11-26T07:06:54.413038Z","shell.execute_reply.started":"2025-11-26T07:06:54.397554Z","shell.execute_reply":"2025-11-26T07:06:54.412326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(len(res_val_df))\n# res_val_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.413817Z","iopub.execute_input":"2025-11-26T07:06:54.414118Z","iopub.status.idle":"2025-11-26T07:06:54.426537Z","shell.execute_reply.started":"2025-11-26T07:06:54.414096Z","shell.execute_reply":"2025-11-26T07:06:54.425847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# res_df_sorted = res_val_df.sort_values(['customer_id', 'prediction'], ascending=[True, False])\n\n# top12_recs_submit = res_df_sorted.groupby('customer_id')['article_id'].apply(list).apply(lambda x: x[:12]).reset_index()\n# top12_recs_submit.rename(columns={'article_id': 'recommendations'}, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.427283Z","iopub.execute_input":"2025-11-26T07:06:54.427537Z","iopub.status.idle":"2025-11-26T07:06:54.439903Z","shell.execute_reply.started":"2025-11-26T07:06:54.427518Z","shell.execute_reply":"2025-11-26T07:06:54.439217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(len(top12_recs_submit))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.440556Z","iopub.execute_input":"2025-11-26T07:06:54.440802Z","iopub.status.idle":"2025-11-26T07:06:54.454505Z","shell.execute_reply.started":"2025-11-26T07:06:54.440781Z","shell.execute_reply":"2025-11-26T07:06:54.453861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =============================================================================\n# # 7. TẠO FILE SUBMISSION\n# # =============================================================================\n# print(\"\\nBắt đầu tạo file submission...\")\n\n# # Lấy DataFrame chứa top 12 gợi ý đã tính toán ở bước trước\n# # DataFrame này có cột 'customer_id' và 'recommendations' (dạng list)\n# submission_df = top12_recs_submit.copy()\n\n# # Chuyển đổi cột list các article_id thành một chuỗi duy nhất, phân tách bằng dấu cách\n# submission_df['prediction'] = submission_df['recommendations'].apply(lambda x: ' '.join(x))\n\n# # Giữ lại các cột cần thiết theo định dạng yêu cầu\n# submission_df = submission_df[['customer_id', 'prediction']]\n\n# # Xuất ra file CSV\n# # index=False là rất quan trọng để không ghi thêm cột chỉ số của DataFrame\n# submission_df.to_csv('submission.csv', index=False)\n\n# print(\"Đã tạo file 'submission.csv' thành công.\")\n# print(\"Xem trước 5 dòng đầu của file submission:\")\n# print(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T07:06:54.455239Z","iopub.execute_input":"2025-11-26T07:06:54.455470Z","iopub.status.idle":"2025-11-26T07:06:54.467294Z","shell.execute_reply.started":"2025-11-26T07:06:54.455456Z","shell.execute_reply":"2025-11-26T07:06:54.466785Z"}},"outputs":[],"execution_count":null}]}