{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":693458,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":525857,"modelId":539917}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# 1. Install Libraries\n!pip install -U sentence-transformers rank_bm25 faiss-cpu textstat\n\n# 2. Imports\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport spacy\nimport re\nimport faiss\nfrom collections import Counter\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom rank_bm25 import BM25Okapi\nfrom sentence_transformers import SentenceTransformer\n\n# 3. Configs\nimport warnings\nwarnings.filterwarnings('ignore')\nsns.set_style(\"whitegrid\")\nplt.rcParams.update({'font.size': 11, 'figure.dpi': 120})\n\n# 4. Load Spacy\ntry:\n    nlp = spacy.load(\"en_core_web_sm\", disable=['parser', 'ner'])\nexcept:\n    !python -m spacy download en_core_web_sm\n    nlp = spacy.load(\"en_core_web_sm\", disable=['parser', 'ner'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:14:53.229873Z","iopub.execute_input":"2025-12-22T06:14:53.230049Z","iopub.status.idle":"2025-12-22T06:15:35.381351Z","shell.execute_reply.started":"2025-12-22T06:14:53.230030Z","shell.execute_reply":"2025-12-22T06:15:35.380547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Load Data\ndf = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv', dtype=str)\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:16:34.199253Z","iopub.execute_input":"2025-12-22T06:16:34.199593Z","iopub.status.idle":"2025-12-22T06:16:35.023568Z","shell.execute_reply.started":"2025-12-22T06:16:34.199568Z","shell.execute_reply":"2025-12-22T06:16:35.022836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:16:42.179591Z","iopub.execute_input":"2025-12-22T06:16:42.180188Z","iopub.status.idle":"2025-12-22T06:16:42.312051Z","shell.execute_reply.started":"2025-12-22T06:16:42.180160Z","shell.execute_reply":"2025-12-22T06:16:42.311469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(\"TÌNH TRẠNG DỮ LIỆU KHUYẾT\")\nmissing_desc = df['detail_desc'].isna() | (df['detail_desc'] == '')\nmissing_count = missing_desc.sum()\nmissing_pct = (missing_count / len(df)) * 100\n\n# Vẽ biểu đồ tròn thể hiện tỷ lệ Missing\nplt.figure(figsize=(6, 6))\nplt.pie([missing_count, len(df)-missing_count], \n        labels=[f'Missing ({missing_pct:.2f}%)', 'Available'], \n        colors=['#E74C3C', '#2ECC71'], explode=(0.1, 0), autopct='%1.1f%%', startangle=90)\nplt.title(f\"Tỷ lệ sản phẩm bị thiếu mô tả (detail_desc)\")\nplt.show()\n\nprint(\"---  VẼ 0.2: GAP ANALYSIS (KHOẢNG TRỐNG THÔNG TIN) ---\")\n\ndef check_category_in_desc(row):\n    # Kiểm tra xem tên loại sản phẩm (VD: Shoes) có nằm trong mô tả không?\n    if pd.isna(row['detail_desc']) or pd.isna(row['product_type_name']):\n        return False\n    return str(row['product_type_name']).lower() in str(row['detail_desc']).lower()\n\ndf['has_category_in_desc'] = df.apply(check_category_in_desc, axis=1)\nmissing_keyword_count = (~df['has_category_in_desc']).sum()\n\n# Vẽ Barplot so sánh\nplt.figure(figsize=(8, 5))\nsns.countplot(x='has_category_in_desc', data=df, palette='viridis')\nplt.title(\"Sản phẩm có chứa Tên Category trong Mô tả không?\")\nplt.xticks([0, 1], [f'NO (Cần Enrichment)\\n{missing_keyword_count} sp', 'YES (Đủ thông tin)'])\nplt.ylabel(\"Số lượng sản phẩm\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:17:07.039953Z","iopub.execute_input":"2025-12-22T06:17:07.040536Z","iopub.status.idle":"2025-12-22T06:17:08.473017Z","shell.execute_reply.started":"2025-12-22T06:17:07.040511Z","shell.execute_reply":"2025-12-22T06:17:08.472247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cấu hình để hiển thị full nội dung mô tả (không bị cắt bớt ...)\npd.set_option('display.max_colwidth', None)\n\n# 1. Lọc ra các dòng \"có vấn đề\" (Problematic Rows)\n# Là những dòng mà has_category_in_desc == False (từ bước trước)\nproblem_rows = df[df['has_category_in_desc'] == False]\n\n# 2. Chọn lọc các cột cần hiển thị để so sánh\ncols_evidence = ['article_id', 'prod_name','colour_group_name', 'product_type_name', 'detail_desc']\n\nprint(f\"--- BẰNG CHỨNG THỰC TẾ: {len(problem_rows)} sản phẩm thiếu từ khóa phân loại ---\")\n\n# 3. Lấy mẫu ngẫu nhiên 5 dòng để kiểm chứng\n# random_state=42 để đảm bảo lần nào chạy cũng ra kết quả giống nhau (dễ viết báo cáo)\nsample_evidence = problem_rows[cols_evidence].sample(5, random_state=42)\ndisplay(sample_evidence)\n\nprint(\"\\n--- KIỂM TRA CỤ THỂ NHÓM GIÀY (SHOES) ---\")\nshoes_issues = problem_rows[\n    problem_rows['product_type_name'].astype(str).str.contains('Shoe|Sneaker', case=False)\n]\n\nif not shoes_issues.empty:\n    display(shoes_issues[cols_evidence].head(3))\nelse:\n    print(\"Nhóm giày dữ liệu tốt, không tìm thấy lỗi.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:17:11.281506Z","iopub.execute_input":"2025-12-22T06:17:11.281804Z","iopub.status.idle":"2025-12-22T06:17:11.356049Z","shell.execute_reply.started":"2025-12-22T06:17:11.281781Z","shell.execute_reply":"2025-12-22T06:17:11.355153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Kiểm tra trùng lặp hoàn toàn (Giống nhau y hệt ở TẤT CẢ các cột)\nnum_full_duplicates = df.duplicated().sum()\nprint(f\"Tổng số dòng trùng lặp 100% (Full Duplicates): {num_full_duplicates}\")\n\nif num_full_duplicates > 0:\n    print(\"Mẫu các dòng trùng lặp:\")\n    # keep=False để hiện tất cả các bản sao ra\n    display(df[df.duplicated(keep=False)].sort_values(by=['article_id']).head())\nelse:\n    print(\"=> Dữ liệu sạch, không có dòng nào trùng lặp hoàn toàn.\")\n\nprint(\"-\" * 30)\n\n# 2. Kiểm tra trùng lặp Article ID (Quan trọng hơn: 1 ID có bị lặp lại 2 lần không?)\nnum_id_duplicates = df.duplicated(subset=['article_id']).sum()\nprint(f\"Tổng số ID bị trùng (Duplicate Primary Key): {num_id_duplicates}\")\n\nif num_id_duplicates > 0:\n    print(\"Cảnh báo: Có ID xuất hiện nhiều lần trong dataset!\")\n    display(df[df.duplicated(subset=['article_id'], keep=False)].sort_values(by=['article_id']).head())\nelse:\n    print(\"=> ID là duy nhất (Unique), cấu trúc chuẩn.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:17:14.169675Z","iopub.execute_input":"2025-12-22T06:17:14.169971Z","iopub.status.idle":"2025-12-22T06:17:14.361081Z","shell.execute_reply.started":"2025-12-22T06:17:14.169946Z","shell.execute_reply":"2025-12-22T06:17:14.360473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- VẼ 1: Text Length Distribution ---\nprint(\"📊 [PLOT 1/7] Text Length Distribution\")\nraw_lens = df['detail_desc'].apply(lambda x: len(str(x).split()))\n\nplt.figure(figsize=(10, 4))\nsns.histplot(raw_lens, bins=40, kde=True, color='#2E86C1')\nplt.axvline(raw_lens.mean(), color='red', linestyle='--', label=f'Mean: {raw_lens.mean():.1f}')\nplt.title(\"Phân bố độ dài mô tả (Tokens)\")\nplt.xlabel(\"Số từ\"); plt.ylabel(\"Số lượng sản phẩm\")\nplt.legend()\nplt.show()\nprint(\"📊 [PLOT 2/7] POS Distribution (Raw)\")\n\nsample_docs = (\n    df['detail_desc']\n    .dropna()                 #\n    .astype(str)              \n    .sample(n=min(2000, df['detail_desc'].notna().sum()), random_state=42)\n    .tolist()\n)\n\npos_counter = Counter()\nfor doc in nlp.pipe(sample_docs, batch_size=64):\n    pos_counter.update([t.pos_ for t in doc])\n\npos_df = (pd.DataFrame.from_dict(pos_counter, orient='index', columns=['Count'])\n          .sort_values('Count', ascending=False)\n          .head(10))\n\nplt.figure(figsize=(10, 4))\nsns.barplot(x=pos_df['Count'], y=pos_df.index, palette='magma')\nplt.title(\"Top 10 Từ loại (POS Tags)\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:17:16.238542Z","iopub.execute_input":"2025-12-22T06:17:16.239188Z","iopub.status.idle":"2025-12-22T06:17:20.960580Z","shell.execute_reply.started":"2025-12-22T06:17:16.239155Z","shell.execute_reply":"2025-12-22T06:17:20.959877Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Biểu đồ 1: Phân bố độ dài mô tả (Text Length Distribution)\nQuan sát dữ liệu: Độ dài trung bình của mô tả sản phẩm rơi vào khoảng 23.8 từ/mẫu. Phân phối lệch phải, tập trung chủ yếu ở dải 15-40 từ.\n\nGiả định & Chiến lược:\n\nDữ liệu thuộc dạng Short-text (Văn bản ngắn). Ta có giả định rằng mô tả sản phẩm thời trang thường là thuộc tính của sản phẩm một cách súc tích thay vì văn bản kể chuyện dài dòng.\n\nĐộ dài này nằm trong \"vùng tối ưu\" của mô hình SBERT (vốn hoạt động tốt nhất với câu < 128 tokens). Chúng ta không cần áp dụng các kỹ thuật cắt gọt (truncation) phức tạp mà có thể đưa trực tiếp vào mô hình để trích xuất đặc trưng ngữ nghĩa.","metadata":{}},{"cell_type":"markdown","source":"Biểu đồ 2: Phân bố Từ loại (POS Distribution) <br>\n- Giả định nghiệp vụ: Trong thương mại điện tử thời trang, hành vi tìm kiếm của người dùng (User Search Intent) thường xoay quanh 3 trục chính: WHAT (Là cái gì?), WHO (Của ai/Cho ai?), và ATTRIBUTES (Như thế nào?).\n- WHAT & WHO $\\rightarrow$ NOUN/PROPN: Người dùng tìm tên loại sản phẩm (\"Shoes\", \"Dress\"). Biểu đồ cho thấy NOUN chiếm tỷ trọng áp đảo (Top 1), xác nhận dữ liệu rất giàu thông tin định danh.\n- ATTRIBUTES $\\rightarrow$ ADJ & NUM: Người dùng lọc theo tính chất (\"Black\", \"Slim fit\") hoặc thông số (\"100%\", \"Size S\"). Đây là lý do ta bắt buộc phải giữ lại Tính từ và Số từ.Lý giải việc giữ lại\n- VERB (VBG/VBN Strategy):Mặc dù VERB (Động từ) xuất hiện ở Top 6, nhưng đa số là động từ hành động rác (\"wash\", \"dry\"). Tuy nhiên, nhóm quyết định giữ lại động từ ở dạng VBG (Gerund) và VBN (Past Participle) vì:\n- Công năng: Các từ như \"Running\" (trong Running shoes), \"Swimming\" (trong Swimming suit) tuy là động từ nhưng đóng vai trò định nghĩa công dụng sản phẩm. Nếu loại bỏ, \"Giày chạy bộ\" sẽ chỉ còn là \"Giày\", làm mất đi ý định mua hàng thể thao của user.\n- Gia công: Các từ như \"Knitted\" (Dệt kim), \"Printed\" (In hình), \"Coated\" (Tráng phủ) định nghĩa chất lượng và kiểu dáng gia công. Đây là những từ khóa \"đắt giá\"  giúp phân biệt các dòng sản phẩm cao cấp.\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# 1. Lọc ra các sản phẩm thuộc các nhóm họa tiết quan trọng\ntarget_patterns = ['Stripe', 'Check', 'Denim']\ndf_pattern = df[df['graphical_appearance_name'].isin(target_patterns)].copy()\n\n# 2. Tạo cột kiểm tra: Từ khóa họa tiết có nằm trong Tên hoặc Mô tả không?\n# Logic: Nếu 'Stripe' xuất hiện trong 'prod_name' hoặc 'detail_desc' -> True (Dư thừa)\ndef check_redundancy(row):\n    text = (str(row['prod_name']) + \" \" + str(row['detail_desc'])).lower()\n    pattern = str(row['graphical_appearance_name']).lower()\n    return pattern in text\n\ndf_pattern['is_redundant'] = df_pattern.apply(check_redundancy, axis=1)\n\n# 3. VẼ BIỂU ĐỒ (VISUALIZATION)\nplt.figure(figsize=(10, 6))\n# Vẽ biểu đồ cột chồng hoặc nhóm để so sánh\nax = sns.countplot(x='graphical_appearance_name', hue='is_redundant', data=df_pattern, palette='viridis')\n\nplt.title('Kiểm tra độ dư thừa thông tin: Họa tiết có sẵn trong Mô tả không?')\nplt.xlabel('Loại họa tiết')\nplt.ylabel('Số lượng sản phẩm')\nplt.legend(title='Trùng lặp thông tin', labels=['False (Chưa có - Cần giữ)', 'True (Đã có - Dư thừa)'])\n\n# Thêm số liệu lên đầu cột cho ngầu\nfor container in ax.containers:\n    ax.bar_label(container)\n\nplt.show()\n\n# 4. IN RA BẰNG CHỨNG (EVIDENCE ROWS)\ncols_show = ['article_id', 'prod_name', 'graphical_appearance_name', 'detail_desc']\n\nprint(\"\\n=== CASE 1: DƯ THỪA (REDUNDANT) - Có thể loại bỏ cột Họa tiết ===\")\nprint(\"Ví dụ các dòng mà từ khóa họa tiết ĐÃ XUẤT HIỆN trong mô tả:\")\ndisplay(df_pattern[df_pattern['is_redundant'] == True][cols_show].head(3))\n\nprint(\"\\n=== CASE 2: KHÔNG TRÙNG (UNIQUE) - Mất thông tin nếu loại bỏ ===\")\nprint(\"Ví dụ các dòng mà mô tả KHÔNG HỀ NHẮC ĐẾN họa tiết:\")\ndisplay(df_pattern[df_pattern['is_redundant'] == False][cols_show].head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:17:23.504730Z","iopub.execute_input":"2025-12-22T06:17:23.505038Z","iopub.status.idle":"2025-12-22T06:17:23.857965Z","shell.execute_reply.started":"2025-12-22T06:17:23.505014Z","shell.execute_reply":"2025-12-22T06:17:23.857266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# 1. Thống kê tần suất\ngraphical_counts = df['graphical_appearance_name'].value_counts().reset_index()\ngraphical_counts.columns = ['Pattern', 'Count']\n\n# 2. Phân loại \"Chất lượng từ khóa\" (Manual Labeling for Analysis)\n# Những từ này quá chung chung, người dùng ít khi search chính xác từ này\ngeneric_terms = ['Solid', 'All over pattern', 'Melange', 'Transparent', 'Treatment', 'Other structure']\n\ndef classify_term(term):\n    if term in generic_terms:\n        return 'Generic/Noise (Chung chung)'\n    return 'Specific (Cụ thể)'\n\ngraphical_counts['Type'] = graphical_counts['Pattern'].apply(classify_term)\n\n# Tính tổng tỷ lệ\ntotal_rows = len(df)\nnoise_rows = graphical_counts[graphical_counts['Type'] == 'Generic/Noise (Chung chung)']['Count'].sum()\nnoise_pct = (noise_rows / total_rows) * 100\n\nprint(f\"Tổng số dòng: {total_rows}\")\nprint(f\"Số dòng chứa từ khóa chung chung (Noise): {noise_rows}\")\nprint(f\"Tỷ lệ Nhiễu thông tin: {noise_pct:.2f}%\")\n\n# 3. Vẽ biểu đồ Top 10\nplt.figure(figsize=(12, 6))\n# Lấy top 10 để vẽ\ntop_10 = graphical_counts.head(10)\ncolors = ['#E74C3C' if x == 'Generic/Noise (Chung chung)' else '#2ECC71' for x in top_10['Type']]\n\nbarplot = sns.barplot(x='Count', y='Pattern', data=top_10, palette=colors)\n\n# Thêm chú thích\nplt.title(f'Phân tích chất lượng từ khóa cột Họa tiết (Noise Rate: {noise_pct:.1f}%)')\nplt.xlabel('Số lượng sản phẩm')\nplt.axvline(x=len(df)*0.5, color='gray', linestyle='--', label='50% Dữ liệu')\n\n# Tạo legend thủ công\nfrom matplotlib.patches import Patch\nlegend_elements = [Patch(facecolor='#E74C3C', label='Noise (Từ chung chung - Nên bỏ)'),\n                   Patch(facecolor='#2ECC71', label='Specific (Từ cụ thể - Có giá trị)')]\nplt.legend(handles=legend_elements)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:17:26.149517Z","iopub.execute_input":"2025-12-22T06:17:26.149814Z","iopub.status.idle":"2025-12-22T06:17:26.426199Z","shell.execute_reply.started":"2025-12-22T06:17:26.149791Z","shell.execute_reply":"2025-12-22T06:17:26.425445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef construct_smart_text(row):\n    color = str(row['colour_group_name']).strip()\n    p_type = str(row['product_type_name']).strip()\n    name = str(row['prod_name']).strip()\n    desc = str(row['detail_desc']).strip()\n    \n    # A. Khử trùng lặp (Deduplication)\n    # Nếu Tên đã chứa Loại (VD: \"Vito Derby Shoe\" chứa \"Shoe\") -> Ẩn Loại đi\n    show_type = True\n    if p_type.lower() in name.lower(): show_type = False\n    elif p_type.lower().rstrip('s') in name.lower().split(): show_type = False\n    display_type = p_type if show_type else \"\"\n\n    # B. Làm sạch mô tả (Safe Description Cutting)\n    # Cắt bỏ phần đầu mô tả nếu nó lặp lại tên sản phẩm, NHƯNG chỉ cắt nếu phần còn lại đủ dài\n    if desc.lower().startswith(name.lower()):\n        potential_desc = desc[len(name):].strip().lstrip('.,- ')\n        if len(potential_desc) > 15: desc = potential_desc\n    elif desc.lower().startswith(p_type.lower()):\n        potential_desc = desc[len(p_type):].strip().lstrip('.,- ')\n        if len(potential_desc) > 15: desc = potential_desc\n    \n    desc = desc.lstrip('.,- ')\n    \n    # C. Ghép chuỗi (Ưu tiên SBERT đọc từ trái sang phải)\n    components = [color, display_type, name, desc]\n    # Lọc bỏ rỗng và ghép lại\n    clean_text = \" \".join([c for c in components if c and c.lower() != 'nan'])\n    return re.sub(r'\\s+', ' ', clean_text).strip()\n\nprint(\" Đang xây dựng Smart Text (Enrichment)...\")\ndf['rich_source'] = df.apply(construct_smart_text, axis=1)\n\nprint(\" Đã xử lý xong! Ví dụ mẫu:\")\nprint(f\"Gốc: {df.iloc[0]['prod_name']} | {df.iloc[0]['detail_desc']}\")\nprint(f\"Smart: {df.iloc[0]['rich_source']}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:17:29.488465Z","iopub.execute_input":"2025-12-22T06:17:29.488769Z","iopub.status.idle":"2025-12-22T06:17:31.890180Z","shell.execute_reply.started":"2025-12-22T06:17:29.488745Z","shell.execute_reply":"2025-12-22T06:17:31.889549Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ta thấy chuỗi này xác suất cao chứa đủ mọi keywords mà user có thể nghĩ ra: <br>\n- Chất liệu: Jersey\n\n- Kiểu dáng: Strap, Narrow shoulder\n\n- Loại: Vest, Top\n\n- Màu: Black","metadata":{}},{"cell_type":"code","source":"class TextPreprocessor:\n    def __init__(self, nlp_model):\n        self.nlp = nlp_model\n        \n    def process_lexical(self, texts):\n        \"\"\"Pipeline 1: Cho BM25 (Giữ Exact Term, No Lemma, Fix Regex %)\"\"\"\n        # 1. Regex Clean: Giữ chữ, số, -, %\n        cleaned_text = [re.sub(r\"[^a-z0-9\\s\\-\\%]\", \" \", str(t).lower()) for t in texts]\n        cleaned_text = [re.sub(r\"\\s+\", \" \", t).strip() for t in cleaned_text]\n        \n        results = []\n        \n        # LOGIC CHỈ GIỮ VERB DẠNG TÍNH TỪ (VBG/VBN)\n        allowed_pos = {'NOUN', 'ADJ', 'PROPN', 'NUM'} \n        allowed_verb_tags = {'VBG', 'VBN'} \n        \n        for doc in self.nlp.pipe(cleaned_text, batch_size=2000):\n            tokens = []\n            for t in doc:\n                # Bỏ Stopword/Punctuation\n                if t.is_stop or t.is_punct or len(t.text) < 2: continue\n                \n                # Logic lọc: Giữ nếu là POS cho phép HOẶC là Verb đặc biệt\n                is_valid_pos = t.pos_ in allowed_pos\n                is_special_verb = (t.pos_ == 'VERB' and t.tag_ in allowed_verb_tags)\n                \n                if is_valid_pos or is_special_verb:\n                    tokens.append(t.text) # Dùng .text để giữ nguyên dạng từ\n                    \n            results.append(\" \".join(tokens))\n        return results\n        \n    def process_semantic(self, df):\n        return df['rich_source'].tolist() \n\nprint(\" Đang chạy Preprocessing...\")\nprep = TextPreprocessor(nlp)\n\n# Pipeline 1: Lexical chạy trên RICH SOURCE\ndf['clean_lexical'] = prep.process_lexical(df['rich_source'])\n\nprint(\" Preprocessing hoàn tất!\")\nprint(f\"🔹 Lexical Sample: {df['clean_lexical'].iloc[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:17:34.055514Z","iopub.execute_input":"2025-12-22T06:17:34.055815Z","iopub.status.idle":"2025-12-22T06:20:24.756666Z","shell.execute_reply.started":"2025-12-22T06:17:34.055790Z","shell.execute_reply":"2025-12-22T06:20:24.755809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\n\n# --- VẼ 3: Vocabulary Size ---\nprint(\"📊 [PLOT 3/7] Vocabulary Size Comparison\")\n\n# --- FIX LỖI TẠI ĐÂY: Thêm .astype(str) để đảm bảo không bị lỗi float ---\nvocab_raw = set(\" \".join(df['detail_desc'].astype(str)).lower().split())\nvocab_clean = set(\" \".join(df['clean_lexical'].astype(str)).split())\n\nplt.figure(figsize=(6, 4))\nsns.barplot(x=['Raw Vocab', 'Clean Lexical Vocab'], y=[len(vocab_raw), len(vocab_clean)], palette='viridis')\nplt.title(f\"Vocab size trước và sau: {len(vocab_raw)} -> {len(vocab_clean)}\")\nplt.ylabel(\"Số lượng từ vựng (Unique)\")\nfor i, v in enumerate([len(vocab_raw), len(vocab_clean)]):\n    plt.text(i, v, str(v), ha='center', va='bottom', fontweight='bold')\nplt.show()\n\n# --- VẼ 4: Zipf's Law ---\nprint(\"📊 [PLOT 4/7] Token Frequency (Zipf's Curve)\")\n# Cũng thêm .astype(str) cho chắc ăn\nall_tokens = \" \".join(df['clean_lexical'].astype(str)).split()\ncounts = Counter(all_tokens).most_common()\nfreqs = [x[1] for x in counts]\nranks = range(1, len(freqs)+1)\n\nplt.figure(figsize=(8, 5))\nplt.loglog(ranks, freqs, marker='.', linestyle='none', alpha=0.3, color='purple')\nplt.title(\"Zipf's Law Check (Log-Log Scale)\")\nplt.xlabel(\"Rank\"); plt.ylabel(\"Frequency\")\nplt.grid(True, which=\"both\", ls=\"-\", alpha=0.5)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T16:33:53.214394Z","iopub.execute_input":"2025-12-21T16:33:53.214906Z","iopub.status.idle":"2025-12-21T16:33:54.836244Z","shell.execute_reply.started":"2025-12-21T16:33:53.214880Z","shell.execute_reply":"2025-12-21T16:33:54.835509Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Mặc dù bước Preprocessing đã loại bỏ hàng nghìn từ vô nghĩa (stopwords, punctuation).\n\nTuy nhiên, bước Data Enrichment đã bổ sung thêm rất nhiều thực thể quan trọng như từ Tên và Loại sản phẩm (ví dụ: tên dòng giày, màu sắc cụ thể) mà phần mô tả gốc không có","metadata":{}},{"cell_type":"markdown","source":"TRAINNING MODELS","metadata":{}},{"cell_type":"code","source":"# 1. Train BM25\nprint(\" [1/3] Training BM25 Index...\")\ntokenized_lexical = [t.split() for t in df['clean_lexical']]\nbm25 = BM25Okapi(tokenized_lexical)\n\n# 2. Encode SBERT\nprint(\" [2/3] Encoding SBERT Embeddings...\")\n# Dùng Smart Text để encode\nsbert_model = SentenceTransformer('all-MiniLM-L6-v2')\nembeddings = sbert_model.encode(df['rich_source'].tolist(), batch_size=64, show_progress_bar=True)\nembeddings = embeddings.astype('float32')\n\n# 3. Clustering (KMeans)\nprint(\" [3/3] Clustering (For Visualization)...\")\nkmeans = KMeans(n_clusters=8, random_state=42).fit(embeddings)\ndf['cluster'] = kmeans.labels_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T23:02:24.822431Z","iopub.execute_input":"2025-12-21T23:02:24.822943Z","iopub.status.idle":"2025-12-21T23:03:09.398361Z","shell.execute_reply.started":"2025-12-21T23:02:24.822919Z","shell.execute_reply":"2025-12-21T23:03:09.397516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"⏳ Running PCA...\")\npca = PCA(n_components=2)\nreduced_vecs = pca.fit_transform(embeddings)\n\n# --- VẼ 5: Embedding Projection ---\nprint(\"📊 [PLOT 5/7] Semantic Structure\")\nplt.figure(figsize=(10, 6))\nplt.scatter(reduced_vecs[:,0], reduced_vecs[:,1], s=1, alpha=0.3, c='gray')\nplt.title(\"Không gian ngữ nghĩa SBERT (PCA Projection)\")\nplt.xlabel(\"PC1\"); plt.ylabel(\"PC2\")\nplt.show()\n\n# --- VẼ 6: Cluster Visualization ---\nprint(\"📊 [PLOT 6/7] Pattern Discovery (Clustering)\")\nplt.figure(figsize=(10, 6))\nscatter = plt.scatter(reduced_vecs[:,0], reduced_vecs[:,1], c=df['cluster'], cmap='tab10', s=3, alpha=0.7)\nplt.colorbar(scatter, label='Cluster ID')\nplt.title(\"Phân cụm sản phẩm dựa trên ngữ nghĩa\")\nplt.xlabel(\"PC1\"); plt.ylabel(\"PC2\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T16:34:50.658735Z","iopub.execute_input":"2025-12-21T16:34:50.659031Z","iopub.status.idle":"2025-12-21T16:34:53.194129Z","shell.execute_reply.started":"2025-12-21T16:34:50.659008Z","shell.execute_reply":"2025-12-21T16:34:53.192926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class HybridSearchEngine:\n    def __init__(self, df, bm25_model, sbert_model, embeddings):\n        self.df = df.reset_index(drop=True)\n        self.bm25 = bm25_model\n        self.sbert_model = sbert_model\n        self.embeddings = embeddings\n        \n        # FAISS Index\n        faiss.normalize_L2(self.embeddings)\n        self.index = faiss.IndexFlatIP(self.embeddings.shape[1])\n        self.index.add(self.embeddings)\n        \n        # --- PHRASE-BASED SYNONYMS (OPTIMIZED DOMAIN KNOWLEDGE) ---\n        # Đã loại bỏ từ \"shoes\" chung chung để tránh boost nhầm giày tây (Derby)\n        self.phrase_synonyms = {\n            'running shoes': ['trainers', 'sneakers', 'runners', 'athletic footwear'],\n            'running shoe': ['trainers', 'sneakers', 'runners'],\n            'gym shoes': ['trainers', 'sneakers'],\n            'joggers': ['sweatpants', 'track pants'], \n            'denim jeans': ['blue jeans', 'denim'],\n            'hoodie': ['sweatshirt', 'hooded'],\n            'summer dress': ['sundress', 'floral dress']\n        }\n        print(\"✅ Engine Ready: Smart Text + Optimized Phrase Expansion.\")\n\n    def _min_max_normalize(self, scores):\n        min_s, max_s = np.min(scores), np.max(scores)\n        if max_s - min_s == 0: return np.zeros_like(scores)\n        return (scores - min_s) / (max_s - min_s)\n    \n    def _expand_query_phrase(self, query):\n        \"\"\"Mở rộng query dựa trên cụm từ, tránh nhiễu\"\"\"\n        query_lower = str(query).lower()\n        expansion_terms = []\n        for phrase, synonyms in self.phrase_synonyms.items():\n            if phrase in query_lower:\n                expansion_terms.extend(synonyms)\n        if expansion_terms:\n            # Chỉ thêm từ đặc thù (trainers, sneakers), không thêm từ gốc tránh lặp\n            return query_lower + \" \" + \" \".join(list(set(expansion_terms)))\n        return query_lower\n    \n    def search(self, query, top_k=10, alpha=0.5):\n        # 1. Expand Query\n        expanded_q = self._expand_query_phrase(query)\n        \n        # 2. Lexical Search (Dùng Expanded Query để bắt keyword)\n        q_lexical = re.sub(r\"[^a-z0-9\\s\\-\\%]\", \" \", expanded_q).split()\n        bm25_raw = self.bm25.get_scores(q_lexical)\n        bm25_norm = self._min_max_normalize(bm25_raw)\n        \n        # 3. Semantic Search (Dùng Original Query để giữ ngữ cảnh chính xác)\n        q_vec = self.sbert_model.encode([query]).astype('float32')\n        faiss.normalize_L2(q_vec)\n        D, I = self.index.search(q_vec, len(self.df))\n        \n        sbert_raw = np.zeros(len(self.df))\n        sbert_raw[I[0]] = D[0]\n        sbert_norm = self._min_max_normalize(sbert_raw)\n        \n        # 4. Fusion\n        final_scores = (alpha * bm25_norm) + ((1 - alpha) * sbert_norm)\n        \n        # 5. Result\n        top_indices = np.argsort(final_scores)[::-1][:top_k]\n        results = self.df.iloc[top_indices][['prod_name', 'product_type_name', 'colour_group_name', 'rich_source']].copy()\n        \n        results['score'] = final_scores[top_indices]\n        results['expanded_query'] = expanded_q\n        results['bm25'] = bm25_norm[top_indices]\n        results['sbert'] = sbert_norm[top_indices]\n        \n        return results\n\nengine = HybridSearchEngine(df, bm25, sbert_model, embeddings)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T23:03:56.489844Z","iopub.execute_input":"2025-12-21T23:03:56.490144Z","iopub.status.idle":"2025-12-21T23:03:56.761330Z","shell.execute_reply.started":"2025-12-21T23:03:56.490121Z","shell.execute_reply":"2025-12-21T23:03:56.760433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- TEST SEARCH ---\nquery = \"Black running shoes\"\nprint(f\"🔎 TEST QUERY: '{query}'\")\nprint(\"\\n--- Alpha = 0.5 (Hybrid) ---\")\ndisplay(engine.search(query, top_k=5, alpha=0.5))\n\n# --- VẼ 7: Query Visualization ---\nprint(\"\\n📊 [PLOT 7/7] Query vs Results Visualization\")\ndef plot_query(query_text):\n    results = engine.search(query_text, top_k=5, alpha=0.5)\n    res_indices = results.index.tolist()\n    \n    q_vec = sbert_model.encode([query_text])\n    res_vecs = embeddings[res_indices]\n    \n    combined = np.vstack([q_vec, res_vecs])\n    pca_local = PCA(n_components=2).fit_transform(combined)\n    \n    plt.figure(figsize=(8, 6))\n    plt.scatter(pca_local[0,0], pca_local[0,1], c='red', s=300, marker='*', label='Query')\n    plt.text(pca_local[0,0], pca_local[0,1]+0.02, \"QUERY\", color='red', fontweight='bold')\n    plt.scatter(pca_local[1:,0], pca_local[1:,1], c='blue', s=100, label='Results')\n    \n    for i in range(1, len(pca_local)):\n        plt.plot([pca_local[0,0], pca_local[i,0]], [pca_local[0,1], pca_local[i,1]], 'k--', alpha=0.3)\n        plt.text(pca_local[i,0], pca_local[i,1], f\"#{i}\", fontsize=9)\n        \n    plt.title(f\"Query: '{query_text}'\")\n    plt.legend(); plt.show()\n\nplot_query(query)\n\ndef get_related_products(self, article_id, top_k=5):\n        \"\"\"Gợi ý sản phẩm tương tự dựa trên vector\"\"\"\n        try:\n            # Tìm index của sản phẩm trong dataframe\n            idx = self.df[self.df['article_id'].astype(str) == str(article_id)].index[0]\n            \n            # Lấy vector của nó\n            target_vec = self.embeddings[idx].reshape(1, -1).astype('float32')\n            faiss.normalize_L2(target_vec)\n            \n            # Search (Lấy top_k + 1 vì kết quả đầu tiên là chính nó)\n            D, I = self.index.search(target_vec, top_k + 1)\n            \n            # Bỏ qua kết quả đầu tiên (chính nó)\n            related_indices = I[0][1:]\n            related_products = self.df.iloc[related_indices].copy()\n            related_products['score'] = D[0][1:]\n            \n            return related_products\n        except:\n            return None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T23:04:02.974869Z","iopub.execute_input":"2025-12-21T23:04:02.975169Z","iopub.status.idle":"2025-12-21T23:04:03.676110Z","shell.execute_reply.started":"2025-12-21T23:04:02.975145Z","shell.execute_reply":"2025-12-21T23:04:03.675346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nimport os\nimport shutil\n\n# 1. Tạo thư mục chứa model\nMODEL_DIR = 'models_best'\nif os.path.exists(MODEL_DIR):\n    shutil.rmtree(MODEL_DIR) # Xóa cũ nếu có để ghi mới\nos.makedirs(MODEL_DIR)\n\nprint(f\"💾 Đang lưu Artifacts vào thư mục '{MODEL_DIR}'...\")\n\n# 2. Lưu DataFrame (Chứa Smart Text 'rich_source' cực quan trọng)\n# Chỉ lưu các cột cần thiết để nhẹ file\ncols_to_keep = ['article_id', 'prod_name', 'product_type_name', 'colour_group_name', 'detail_desc', 'rich_source']\ndf_export = df[cols_to_keep].copy()\n\nwith open(f'{MODEL_DIR}/df_products.pkl', 'wb') as f:\n    pickle.dump(df_export, f)\nprint(\"✅ Đã lưu DataFrame (df_products.pkl)\")\n\n# 3. Lưu BM25 Model\nwith open(f'{MODEL_DIR}/bm25_model.pkl', 'wb') as f:\n    pickle.dump(bm25, f)\nprint(\"✅ Đã lưu BM25 Model (bm25_model.pkl)\")\n\n# 4. Lưu SBERT Embeddings (Nặng nhất, lưu dạng numpy)\nnp.save(f'{MODEL_DIR}/sbert_embeddings.npy', embeddings)\nprint(\"✅ Đã lưu Embeddings (sbert_embeddings.npy)\")\n\n# 5. Nén lại thành ZIP để dễ tải về máy (Nếu cần)\nshutil.make_archive(MODEL_DIR, 'zip', MODEL_DIR)\nprint(f\"tải file '{MODEL_DIR}.zip' về máy.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T23:04:34.707164Z","iopub.execute_input":"2025-12-21T23:04:34.707860Z","iopub.status.idle":"2025-12-21T23:04:44.171697Z","shell.execute_reply.started":"2025-12-21T23:04:34.707833Z","shell.execute_reply":"2025-12-21T23:04:44.171080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\nimport os\n\n# Đảm bảo file zip đã được tạo\nzip_file = 'hm_images_50k_optimized.zip'\n\nif os.path.exists(zip_file):\n    print(f\"👇 Bấm vào dòng chữ xanh bên dưới để tải '{zip_file}' về máy:\")\n    display(FileLink(zip_file))\nelse:\n    print(\"⚠️ Chưa thấy file zip. Bro chạy cell nén file ở trên chưa?\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T03:04:56.370489Z","iopub.execute_input":"2025-12-21T03:04:56.371249Z","iopub.status.idle":"2025-12-21T03:04:56.378114Z","shell.execute_reply.started":"2025-12-21T03:04:56.371202Z","shell.execute_reply":"2025-12-21T03:04:56.377461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport zipfile\nfrom tqdm import tqdm\nfrom PIL import Image\n\n# 1. Đọc dữ liệu (Sửa đường dẫn file pkl của bro trên Kaggle)\ndf = pd.read_pickle('/kaggle/input/df-products/pytorch/default/1/df_products.pkl')\nsample_ids = df['article_id'].head(100000).astype(str).str.zfill(10).tolist()\n\nbase_path = '/kaggle/input/h-and-m-personalized-fashion-recommendations/images'\noutput_dir = '/kaggle/working/static_images'\nos.makedirs(output_dir, exist_ok=True)\n\nprint(\"🚀 Đang nén ảnh siêu nhỏ...\")\ncount = 0\nfor aid in tqdm(sample_ids):\n    sub_folder = aid[:3]\n    img_path = os.path.join(base_path, sub_folder, f\"{aid}.jpg\")\n    \n    if os.path.exists(img_path):\n        try:\n            with Image.open(img_path) as img:\n                img = img.convert('RGB')\n                img.thumbnail((200, 200)) # Cho ảnh nhỏ lại\n                # Lưu ảnh với chất lượng thấp để tối ưu dung lượng\n                img.save(os.path.join(output_dir, f\"{aid}.jpg\"), optimize=True, quality=60)\n                count += 1\n        except: continue\n    if count >= 10000: break\n\n# 2. Nén thành file ZIP\nzip_name = 'hm_10k_compressed.zip'\nwith zipfile.ZipFile(zip_name, 'w', zipfile.ZIP_DEFLATED) as zipf:\n    for file in os.listdir(output_dir):\n        zipf.write(os.path.join(output_dir, file), arcname=file)\n\nprint(f\"✨ XONG! Tải file này về và up lên HF: {zip_name} (Số lượng: {count})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T02:16:10.642381Z","iopub.status.idle":"2025-12-21T02:16:10.642603Z","shell.execute_reply.started":"2025-12-21T02:16:10.642495Z","shell.execute_reply":"2025-12-21T02:16:10.642509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport zipfile\nfrom tqdm import tqdm\nfrom PIL import Image\nimport shutil\n\n# 1. Đọc dữ liệu và lấy danh sách ID\ndf =  pd.read_pickle('/kaggle/input/df-products/pytorch/default/1/df_products.pkl')\n# Lấy 50,000 ảnh đầu tiên (hoặc bỏ .head() nếu muốn lấy hết - sẽ lâu hơn)\nall_ids = df['article_id'].astype(str).str.zfill(10).unique().tolist()\ntarget_ids = all_ids[:100000] \n\nbase_path = '/kaggle/input/h-and-m-personalized-fashion-recommendations/images'\ntemp_dir = '/kaggle/working/temp_images_compressed'\nos.makedirs(temp_dir, exist_ok=True)\n\nprint(f\"🚀 Đang xử lý và nén {len(target_ids)} ảnh...\")\ncount = 0\nfor aid in tqdm(target_ids):\n    sub_folder = aid[:3]\n    img_path = os.path.join(base_path, sub_folder, f\"{aid}.jpg\")\n    \n    if os.path.exists(img_path):\n        try:\n            # Nén ảnh nhỏ lại để tiết kiệm dung lượng\n            with Image.open(img_path) as img:\n                img = img.convert('RGB')\n                img.thumbnail((250, 250)) # Kích thước đủ xem demo\n                # Lưu vào thư mục tạm, tên file chỉ là ID.jpg\n                img.save(os.path.join(temp_dir, f\"{aid}.jpg\"), optimize=True, quality=65)\n                count += 1\n        except: continue\n\n# 2. Nén thư mục tạm thành file ZIP (Không nén folder cha)\nzip_name = 'hm_images_50k_optimized.zip'\nprint(f\"📦 Đang đóng gói {count} ảnh vào file ZIP...\")\nwith zipfile.ZipFile(zip_name, 'w', zipfile.ZIP_DEFLATED) as zipf:\n    for file in os.listdir(temp_dir):\n        zipf.write(os.path.join(temp_dir, file), arcname=file)\n\n# Dọn dẹp\nshutil.rmtree(temp_dir)\nprint(f\"✨ XONG! Tải file này về: {zip_name}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T02:17:52.178644Z","iopub.execute_input":"2025-12-21T02:17:52.179344Z","iopub.status.idle":"2025-12-21T03:03:00.186034Z","shell.execute_reply.started":"2025-12-21T02:17:52.179315Z","shell.execute_reply":"2025-12-21T03:03:00.185353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Load model (đã load sẵn trong code chính rồi)\n# model = SentenceTransformer('all-MiniLM-L6-v2')\n\n# 2. Chạy thử 1 query\nquery = \"Running shoes\"\nquery_vector = sbert_model.encode([query])\n\n# 3. In ra màn hình\nprint(\"Kích thước vector truy vấn:\", query_vector.shape)\nprint(\"Dữ liệu vector (5 giá trị đầu):\", query_vector[0][:5])\nprint(\"-\" * 20)\nprint(\"Kích thước ma trận embeddings toàn bộ:\", embeddings.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T06:16:21.985751Z","iopub.execute_input":"2025-12-22T06:16:21.986486Z","iopub.status.idle":"2025-12-22T06:16:21.993566Z","shell.execute_reply.started":"2025-12-22T06:16:21.986450Z","shell.execute_reply":"2025-12-22T06:16:21.992600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Xử lý query (Tokenize)\nquery = \"Running shoes\"\ntokenized_query = query.lower().split() # Hoặc dùng hàm preprocess của bạn\n\n# 2. Lấy điểm số BM25\nbm25_scores = bm25.get_scores(tokenized_query)\n\n# 3. In ra\nprint(\"Số lượng điểm số trả về:\", len(bm25_scores))\nprint(\"Điểm số của 10 sản phẩm đầu tiên:\", bm25_scores[:10])\nprint(\"Điểm cao nhất:\", max(bm25_scores))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cài đặt pyspark nếu chưa có\n!pip install pyspark","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:31:15.736179Z","iopub.execute_input":"2025-12-27T14:31:15.736675Z","iopub.status.idle":"2025-12-27T14:31:20.166424Z","shell.execute_reply.started":"2025-12-27T14:31:15.736626Z","shell.execute_reply":"2025-12-27T14:31:20.165640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql.functions import col, count\nfrom pyspark.ml.feature import StringIndexer\nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.ml.evaluation import RegressionEvaluator\n\n# ---------------------------------------------------------\n# BƯỚC 1: KHỞI TẠO SPARK SESSION (LOCAL MODE)\n# ---------------------------------------------------------\n# ---------------------------------------------------------\n# BƯỚC 1: KHỞI TẠO SPARK SESSION (CẤU HÌNH CHO BIG DATA)\n# ---------------------------------------------------------\nspark = SparkSession.builder \\\n    .appName(\"HM_Recommendation_ALS_Local\") \\\n    .config(\"spark.driver.memory\", \"14g\") \\\n    .config(\"spark.executor.memory\", \"14g\") \\\n    .config(\"spark.sql.shuffle.partitions\", \"200\") \\\n    .config(\"spark.kryoserializer.buffer.max\", \"1g\") \\\n    .config(\"spark.driver.maxResultSize\", \"4g\") \\\n    .master(\"local[*]\") \\\n    .getOrCreate()\n\nprint(f\"Spark Version: {spark.version}\")\nprint(\"Đã cấu hình xong Kryo Buffer 1GB\")\n\n# ---------------------------------------------------------\n# BƯỚC 2: ĐỌC VÀ TIỀN XỬ LÝ DỮ LIỆU (DATA PREPROCESSING)\n# ---------------------------------------------------------\n# Đường dẫn dataset trên Kaggle\ndataset_path = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\"\n\n# Đọc dữ liệu. InferSchema=True để Spark tự hiểu kiểu dữ liệu (tốn chút thời gian nhưng tiện)\n# Lưu ý: H&M dataset rất lớn (>30 triệu dòng). Để demo chạy nhanh, thầy sẽ lấy mẫu (sampling)\n# hoặc lọc dữ liệu gần nhất. Ở đây thầy lấy 1 triệu dòng đầu tiên để em test pipeline trước.\ndf_raw = spark.read.csv(dataset_path, header=True, inferSchema=True).limit(100000)\n\n# Chọn các cột cần thiết\ndf = df_raw.select(\"t_dat\", \"customer_id\", \"article_id\")\n\n# --- Feature Engineering: Tạo cột 'rating' ---\n# H&M là bài toán Implicit Feedback (người dùng không chấm 1-5 sao).\n# Ta giả định: Số lần mua 1 sản phẩm = Mức độ quan tâm (Rating).\ndf_rating = df.groupBy(\"customer_id\", \"article_id\") \\\n    .agg(count(\"article_id\").alias(\"rating\"))\n\n# --- StringIndexer ---\n# ALS của Spark yêu cầu input là số nguyên (integer), nhưng customer_id và article_id của H&M là chuỗi hash.\n# Ta phải map chúng sang index số (0, 1, 2...).\n\n# Indexing cho User\nuser_indexer = StringIndexer(inputCol=\"customer_id\", outputCol=\"user_idx\")\nprint(user_indexer)\nuser_indexer_model = user_indexer.fit(df_rating)\nprint(\"user index model: \", user_indexer_model)\ndf_indexed = user_indexer_model.transform(df_rating)\n\n# Indexing cho Item\nitem_indexer = StringIndexer(inputCol=\"article_id\", outputCol=\"item_idx\")\nitem_indexer_model = item_indexer.fit(df_indexed)\ndf_final = item_indexer_model.transform(df_indexed)\n\n# Cache dữ liệu vào RAM để training nhanh hơn\n# df_final.cache()\n\nprint(\"Dữ liệu đã sẵn sàng cho ALS:\")\ndf_final.show(5)\n\n# ---------------------------------------------------------\n# BƯỚC 3: XÂY DỰNG VÀ HUẤN LUYỆN MÔ HÌNH (MODELING)\n# ---------------------------------------------------------\n# Chia tập train/test (80/20)\n(training, test) = df_final.randomSplit([0.8, 0.2])\n\n# Cấu hình ALS\n# - rank: Số lượng factors tiềm ẩn (User Factors & Item Factors).\n# - maxIter: Số vòng lặp tối ưu hóa.\n# - regParam: Tham số regularization để tránh overfitting.\n# - implicitPrefs=True: RẤT QUAN TRỌNG. Báo cho Spark biết đây là dữ liệu hành vi (mua/không mua), không phải chấm điểm.\n# - coldStartStrategy=\"drop\": Bỏ qua các user/item chưa từng xuất hiện trong tập train để tránh lỗi NaN khi predict.\n\nals = ALS(\n    maxIter=5, \n    regParam=0.01, \n    userCol=\"user_idx\", \n    itemCol=\"item_idx\", \n    ratingCol=\"rating\",\n    implicitPrefs=True,\n    coldStartStrategy=\"drop\" \n)\n\nprint(\"Đang huấn luyện mô hình ALS...\")\nmodel = als.fit(training)\nprint(\"Huấn luyện hoàn tất!\")\n\n# ---------------------------------------------------------\n# BƯỚC 4: SINH GỢI Ý (RECOMMENDATION)\n# ---------------------------------------------------------\n\n# Tạo Top-5 gợi ý cho TẤT CẢ user\nprint(\"Đang sinh gợi ý cho users...\")\nuser_recs = model.recommendForAllUsers(5)\n\n# Hiển thị kết quả (Dạng index)\nuser_recs.show(5, truncate=False)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T11:51:55.240206Z","iopub.execute_input":"2025-12-28T11:51:55.240517Z","iopub.status.idle":"2025-12-28T11:53:02.987381Z","shell.execute_reply.started":"2025-12-28T11:51:55.240491Z","shell.execute_reply":"2025-12-28T11:53:02.986671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lấy ra 5 user bất kỳ và xem model gợi ý gì cho họ\nuser_recs = model.recommendForAllUsers(5)\n\nprint(\"--- KẾT QUẢ GỢI Ý MẪU ---\")\nuser_recs.show(5, truncate=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T11:56:10.522909Z","iopub.execute_input":"2025-12-28T11:56:10.523567Z","iopub.status.idle":"2025-12-28T11:56:18.010864Z","shell.execute_reply.started":"2025-12-28T11:56:10.523538Z","shell.execute_reply":"2025-12-28T11:56:18.009740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.ml.evaluation import RankingEvaluator\nfrom pyspark.sql.functions import col, collect_list\n\n# ---------------------------------------------------------\n# BƯỚC 1: CHUẨN BỊ DỮ LIỆU KIỂM THỬ (GROUND TRUTH)\n# ---------------------------------------------------------\nprint(\"Đang chuẩn bị dữ liệu Ground Truth từ tập Test...\")\n\n# SỬA LỖI: Ép kiểu item_idx sang 'double' ngay từ đầu\ntest_casted = test.withColumn(\"item_idx\", col(\"item_idx\").cast(\"double\"))\n\nground_truth = test_casted.groupBy(\"user_idx\") \\\n    .agg(collect_list(\"item_idx\").alias(\"true_items\"))\n\n# ---------------------------------------------------------\n# BƯỚC 2: SINH DỰ ĐOÁN TỪ MODEL (PREDICTIONS)\n# ---------------------------------------------------------\nprint(\"Đang sinh Top-10 gợi ý...\")\nrecommendations = model.recommendForAllUsers(10)\n\n# SỬA LỖI: Ép kiểu mảng dự đoán sang 'array<double>'\nprediction_clean = recommendations.select(\n    \"user_idx\", \n    col(\"recommendations.item_idx\").cast(\"array<double>\").alias(\"predicted_items\")\n)\n\n# ---------------------------------------------------------\n# BƯỚC 3: KẾT HỢP (JOIN) DỰ ĐOÁN VÀ THỰC TẾ\n# ---------------------------------------------------------\neval_df = prediction_clean.join(ground_truth, on=\"user_idx\")\n\n# Cache lại để Spark không phải tính toán lại khi chạy 2 metric\neval_df.cache()\n\nprint(\"Dữ liệu sẵn sàng để đánh giá (Đã convert sang Double):\")\neval_df.printSchema() # In ra để em kiểm tra xem nó thành double chưa\neval_df.show(3, truncate=True)\n\n# ---------------------------------------------------------\n# BƯỚC 4: TÍNH TOÁN METRIC (MAP@10 và NDCG@10)\n# ---------------------------------------------------------\n\nevaluator_map = RankingEvaluator(\n    predictionCol=\"predicted_items\", \n    labelCol=\"true_items\", \n    metricName=\"meanAveragePrecision\", \n    k=10\n)\n\nevaluator_ndcg = RankingEvaluator(\n    predictionCol=\"predicted_items\", \n    labelCol=\"true_items\", \n    metricName=\"ndcgAtK\", \n    k=10\n)\n\nprint(\"Đang tính toán Metric...\")\nmap_score = evaluator_map.evaluate(eval_df)\nndcg_score = evaluator_ndcg.evaluate(eval_df)\n\nprint(\"=\"*40)\nprint(f\"📊 KẾT QUẢ ĐÁNH GIÁ MÔ HÌNH (TOP-10)\")\nprint(f\"✅ MAP@10  : {map_score:.4f}\")\nprint(f\"✅ NDCG@10 : {ndcg_score:.4f}\")\nprint(\"=\"*40)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T11:59:02.871159Z","iopub.execute_input":"2025-12-28T11:59:02.871466Z","iopub.status.idle":"2025-12-28T11:59:24.842382Z","shell.execute_reply.started":"2025-12-28T11:59:02.871434Z","shell.execute_reply":"2025-12-28T11:59:24.841636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql.functions import concat_ws, lit\n\n# 1. Tính toán 12 món được mua nhiều nhất trong 3 tháng qua\n# (Dùng lại df_rating hoặc df_recent của em)\ntop12_items = df_rating.groupBy(\"article_id\").count() \\\n    .orderBy(col(\"count\").desc()) \\\n    .limit(12) \\\n    .select(\"article_id\") \\\n    .collect()\n\n# 2. Chuyển thành chuỗi: \"0751471001 0573085028 ...\"\n# Lưu ý: Thêm số '0' ở đầu nếu ID bị mất số 0\ndef format_id(aid):\n    s = str(aid)\n    return \"0\" * (10 - len(s)) + s\n\ntop12_string = \" \".join([format_id(row.article_id) for row in top12_items])\nprint(f\"Chuỗi mặc định (Best Sellers): {top12_string}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T12:01:49.265775Z","iopub.execute_input":"2025-12-28T12:01:49.266102Z","iopub.status.idle":"2025-12-28T12:01:52.440600Z","shell.execute_reply.started":"2025-12-28T12:01:49.266074Z","shell.execute_reply":"2025-12-28T12:01:52.439815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pyspark.sql.functions import udf\nfrom pyspark.sql.types import StringType\n\n# 1. Lấy danh sách nhãn (Labels) từ Indexer\n# Đây là danh sách [Item0, Item1, Item2...]\nitem_labels = item_indexer_model.labels\nuser_labels = user_indexer_model.labels\n\n# Broadcast danh sách Item để các Worker đọc được\nbc_item_labels = spark.sparkContext.broadcast(item_labels)\nbc_user_labels = spark.sparkContext.broadcast(user_labels)\n\n# 2. Định nghĩa hàm Map ngược (UDF)\ndef map_ids_and_format(indices):\n    # indices: list các item_idx [0.0, 5.0, 10.0...]\n    if not indices: return \"\"\n    \n    result = []\n    labels = bc_item_labels.value\n    for idx in indices:\n        try:\n            # Lấy ID gốc từ danh sách labels\n            original_id = str(labels[int(idx)])\n            # Format lại cho đủ 10 ký tự (thêm số 0 đầu)\n            formatted_id = \"0\" * (10 - len(original_id)) + original_id\n            result.append(formatted_id)\n        except:\n            continue\n    return \" \".join(result) # Nối lại bằng dấu cách\n\n# Đăng ký UDF với Spark\nformat_pred_udf = udf(map_ids_and_format, StringType())\n\n# 3. Áp dụng UDF vào DataFrame kết quả\n# recommendations: [user_idx, [(item_idx, score), ...]]\n# Ta lấy Top 12 luôn cho đúng chuẩn cuộc thi\nrecs_top12 = model.recommendForAllUsers(12)\n\n# Map Item IDX -> Chuỗi Article ID\ndf_preds = recs_top12.select(\n    col(\"user_idx\"),\n    format_pred_udf(col(\"recommendations.item_idx\")).alias(\"prediction\")\n)\n\n# Map User IDX -> Customer ID thật\n# Vì User Labels quá lớn, ta không dùng UDF mà dùng IndexToString sẽ an toàn hơn cho Driver\nfrom pyspark.ml.feature import IndexToString\n\nuser_converter = IndexToString(inputCol=\"user_idx\", outputCol=\"customer_id\", labels=user_labels)\ndf_preds_final = user_converter.transform(df_preds).select(\"customer_id\", \"prediction\")\n\nprint(\"Kết quả dự đoán sau khi format:\")\ndf_preds_final.show(3, truncate=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T12:02:02.396840Z","iopub.execute_input":"2025-12-28T12:02:02.397161Z","iopub.status.idle":"2025-12-28T12:02:16.307057Z","shell.execute_reply.started":"2025-12-28T12:02:02.397134Z","shell.execute_reply":"2025-12-28T12:02:16.306234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------------------------------------\n# BƯỚC 3: JOIN VỚI SAMPLE SUBMISSION (ĐÃ SỬA LỖI AMBIGUOUS)\n# ---------------------------------------------------------\nfrom pyspark.sql.functions import col, lit, coalesce\n\n# 1. Đọc file mẫu (Sample Submission)\n# Ta chỉ cần lấy cột 'customer_id' để đảm bảo đủ user, bỏ qua cột 'prediction' rác trong đó\npath_sample = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\"\ndf_sample_users = spark.read.csv(path_sample, header=True).select(\"customer_id\")\n\n# 2. Đổi tên cột dự đoán của Model để tránh trùng tên (Quan trọng!)\ndf_preds_renamed = df_preds_final.withColumnRenamed(\"prediction\", \"model_prediction\")\n\n# 3. Left Join\n# Logic: Lấy danh sách user chuẩn từ file mẫu, ghép với dự đoán của model\ndf_joined = df_sample_users.join(df_preds_renamed, on=\"customer_id\", how=\"left\")\n\n# 4. Điền vào chỗ trống (Fillna/Coalesce)\n# Nếu model_prediction có dữ liệu -> Dùng nó\n# Nếu model_prediction là null (Cold Start) -> Dùng chuỗi top12_string\ndf_submission = df_joined.select(\n    col(\"customer_id\"),\n    coalesce(col(\"model_prediction\"), lit(top12_string)).alias(\"prediction\")\n)\n\n# 5. Lưu file\nprint(\"Đang lưu file submission.csv (Có thể mất vài phút)...\")\n# coalesce(1): Gom về 1 file duy nhất để dễ download\ndf_submission.coalesce(1).write.csv(\"submission_output\", header=True, mode=\"overwrite\")\n\nprint(\"✅ Đã xong! Hãy vào folder 'submission_output' tải file csv về và nộp.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T12:03:29.554326Z","iopub.execute_input":"2025-12-28T12:03:29.554627Z","iopub.status.idle":"2025-12-28T12:03:45.673080Z","shell.execute_reply.started":"2025-12-28T12:03:29.554603Z","shell.execute_reply":"2025-12-28T12:03:45.671205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom IPython.display import FileLink\n\nprint(\"⏳ Đang nén dữ liệu... (Vui lòng đợi vài phút)\")\n\n# 1. Định nghĩa tên file zip đầu ra\noutput_filename = \"HM_Project_Artifacts\"\nsource_dir = \"./\"  # Thư mục hiện tại (Working Directory)\n\n# 2. Tạo thư mục tạm để gom những thứ cần thiết\ntemp_folder = \"pack_to_download\"\nos.makedirs(temp_folder, exist_ok=True)\n\n# Danh sách các mục quan trọng cần tải về\ntargets = [\n    \"als_hm_model_final\",    # Model Spark (Quan trọng nhất cho Docker)\n    \"user_labels.csv\",       # Map User ID\n    \"item_labels.csv\",       # Map Item ID\n    \"submission_output\"      # Folder chứa file nộp bài\n]\n\n# Copy vào thư mục tạm\nfor t in targets:\n    if os.path.exists(t):\n        # Nếu là thư mục (Model, Submission)\n        if os.path.isdir(t):\n            # Xóa nếu đã tồn tại trong temp để tránh lỗi\n            if os.path.exists(f\"{temp_folder}/{t}\"):\n                shutil.rmtree(f\"{temp_folder}/{t}\")\n            shutil.copytree(t, f\"{temp_folder}/{t}\")\n        # Nếu là file (CSV)\n        else:\n            shutil.copy(t, f\"{temp_folder}/{t}\")\n    else:\n        print(f\"⚠️ Cảnh báo: Không tìm thấy {t}, sẽ bỏ qua.\")\n\n# 3. Nén thư mục tạm thành file ZIP\nshutil.make_archive(output_filename, 'zip', temp_folder)\n\n# 4. Xóa thư mục tạm cho sạch rác\nshutil.rmtree(temp_folder)\n\nprint(f\"✅ Đã nén xong! File của em tên là: {output_filename}.zip\")\nprint(\"👇 Bấm vào link dưới đây để tải về ngay:\")\ndisplay(FileLink(f'{output_filename}.zip'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T12:04:51.842744Z","iopub.execute_input":"2025-12-28T12:04:51.843535Z","iopub.status.idle":"2025-12-28T12:04:57.515334Z","shell.execute_reply.started":"2025-12-28T12:04:51.843503Z","shell.execute_reply":"2025-12-28T12:04:57.514729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_path = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\"\n\n# Đọc dữ liệu. InferSchema=True để Spark tự hiểu kiểu dữ liệu (tốn chút thời gian nhưng tiện)\n# Lưu ý: H&M dataset rất lớn (>30 triệu dòng). Để demo chạy nhanh, thầy sẽ lấy mẫu (sampling)\n# hoặc lọc dữ liệu gần nhất. Ở đây thầy lấy 1 triệu dòng đầu tiên để em test pipeline trước.\ndf_raw = spark.read.csv(dataset_path, header=True, inferSchema=True)\ndf_raw.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T09:59:13.200939Z","iopub.execute_input":"2025-12-28T09:59:13.201339Z","iopub.status.idle":"2025-12-28T09:59:56.873478Z","shell.execute_reply.started":"2025-12-28T09:59:13.201309Z","shell.execute_reply":"2025-12-28T09:59:56.872504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T09:45:23.505023Z","iopub.execute_input":"2025-12-28T09:45:23.505820Z","iopub.status.idle":"2025-12-28T09:45:23.626863Z","shell.execute_reply.started":"2025-12-28T09:45:23.505790Z","shell.execute_reply":"2025-12-28T09:45:23.625191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}