{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":13713247,"sourceType":"datasetVersion","datasetId":8674631}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# =============================================================================\n# === NOTEBOOK #2: TẢI BẢN ĐỒ TỪ DATASET VÀ TẠO OUTPUT THÔ ===\n# =============================================================================\n# Mục tiêu: Tải các bản đồ .pkl (có score) từ bộ dữ liệu Kaggle đã tạo sẵn và\n# tạo ra các file output chi tiết (customer_id, article_id, score).\n# =============================================================================\n\n# =============================================================================\n# === BƯỚC 1: CHUẨN BỊ MÔI TRƯỜNG VÀ DỮ LIỆU ===\n# =============================================================================\nprint(\"Đang chuẩn bị môi trường...\")\n!pip install -q tqdm\nprint(\"Chuẩn bị hoàn tất.\")\n\n# Import các công cụ cần thiết\nimport pandas as pd\nimport numpy as np\nfrom datetime import timedelta\nfrom collections import defaultdict\nimport warnings\nfrom tqdm.auto import tqdm\nimport pickle # Công cụ để \"hồi sinh\" các \"bộ não\" đã lưu\n\nwarnings.filterwarnings('ignore')\nprint(\"Các công cụ đã sẵn sàng.\")\n\n# Tải dữ liệu gốc\nprint(\"\\nĐang tải dữ liệu gốc...\")\ntry:\n    transactions = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', dtype={'article_id': str}, parse_dates=['t_dat'])\n    sample_submission = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\n    print(\"Dữ liệu gốc đã được tải thành công.\")\nexcept FileNotFoundError:\n    print(\"LỖI: Không tìm thấy dữ liệu gốc! Vui lòng thêm bộ dữ liệu 'h-and-m-personalized-fashion-recommendations'.\")\n    exit()\n\n# =============================================================================\n# === BƯỚC 2: TẢI CÁC BẢN ĐỒ RECALL (CÓ SCORE) TỪ DATASET CỦA BẠN ===\n# =============================================================================\nprint(\"\\nĐang tải các 'bộ não' gợi ý (kèm điểm số) từ dataset của bạn...\")\n\n# --- Đường dẫn đã được cập nhật theo thông tin bạn cung cấp ---\npath_to_maps = '/kaggle/input/my-recall-maps/'\n\ntry:\n    # Tải \"bộ não\" Giao dịch\n    with open(f'{path_to_maps}/transaction_map_with_scores.pkl', 'rb') as f:\n        transaction_map = pickle.load(f)\n    print(f\"-> Tải thành công 'bộ não' Giao dịch (chứa {len(transaction_map)} sản phẩm).\")\n\n    # Tải \"bộ não\" Tên sản phẩm\n    with open(f'{path_to_maps}/name_sim_map_with_scores.pkl', 'rb') as f:\n        name_sim_map = pickle.load(f)\n    print(f\"-> Tải thành công 'bộ não' Tên sản phẩm (chứa {len(name_sim_map)} sản phẩm).\")\n\n    # Tải \"bộ não\" Mô tả sản phẩm\n    with open(f'{path_to_maps}/embedding_sim_map_with_scores.pkl', 'rb') as f:\n        embedding_sim_map = pickle.load(f)\n    print(f\"-> Tải thành công 'bộ não' Mô tả sản phẩm (chứa {len(embedding_sim_map)} sản phẩm).\")\n\nexcept FileNotFoundError:\n    print(f\"\\nLỖI: Không tìm thấy các file .pkl trong đường dẫn '{path_to_maps}'!\")\n    print(\"Vui lòng kiểm tra lại xem bạn đã thêm đúng bộ dữ liệu 'my-recall-maps' chưa.\")\n    transaction_map, name_sim_map, embedding_sim_map = {}, {}, {}\n\n# =============================================================================\n# === BƯỚC 3: CÔNG THỨC TẠO OUTPUT RECALL THÔ (TỐI ƯU BỘ NHỚ) ===\n# =============================================================================\ndef generate_raw_recall_output(customers_to_predict, all_transactions, brain_maps: dict, filename: str, recent_item_count=5):\n    print(f\"\\n--- Bắt đầu tạo file output thô: '{filename}' ---\")\n    \n    final_candidates_list = []\n    customer_ids = customers_to_predict['customer_id'].unique()\n    \n    print(\"Chuẩn bị lịch sử mua sắm để tra cứu nhanh...\")\n    full_history_map_sets = all_transactions.groupby('customer_id')['article_id'].apply(set).to_dict()\n    recent_transactions = all_transactions[all_transactions['t_dat'] >= (all_transactions['t_dat'].max() - timedelta(days=30))]\n    recent_history_map = recent_transactions.groupby('customer_id')['article_id'].apply(lambda x: list(dict.fromkeys(x))).to_dict()\n\n    for customer_id in tqdm(customer_ids, desc=f\"Đang xử lý khách hàng cho {filename}\"):\n        \n        seed_items = recent_history_map.get(customer_id, [])[-recent_item_count:]\n        candidate_scores = defaultdict(float)\n        \n        if seed_items:\n            for item in seed_items:\n                for brain_name, brain_map in brain_maps.items():\n                    suggestions_with_scores = brain_map.get(item, [])\n                    for suggested_item, score in suggestions_with_scores:\n                        if score > candidate_scores[suggested_item]:\n                            candidate_scores[suggested_item] = score\n        \n        already_owned_items = full_history_map_sets.get(customer_id, set())\n        \n        for article_id, score in candidate_scores.items():\n            if article_id not in already_owned_items:\n                final_candidates_list.append({\n                    'customer_id': customer_id,\n                    'article_id': article_id,\n                    'score': score\n                })\n            \n    output_df = pd.DataFrame(final_candidates_list)\n    \n    output_df.to_csv(f\"/kaggle/working/{filename}\", index=False)\n    \n    print(f\"Đã tạo thành công file '{filename}' với {len(output_df):,} dòng.\")\n    print(\"Xem trước 5 dòng đầu tiên:\")\n    print(output_df.head())\n\n# =============================================================================\n# === BƯỚC 4: TẠO 3 FILE OUTPUT THÔ RIÊNG BIỆT ===\n# =============================================================================\ntarget_customers = sample_submission\n\n# --- FILE 1: OUTPUT THÔ TỪ GIAO DỊCH ---\ngenerate_raw_recall_output(\n    customers_to_predict=target_customers,\n    all_transactions=transactions,\n    brain_maps={'transactional_score': transaction_map},\n    filename=\"raw_recall_transactional_with_scores.csv\"\n)\n\n# --- FILE 2: OUTPUT THÔ TỪ NỘI DUNG ---\ncontent_brains = {\n    'name_sim_score': name_sim_map,\n    'embedding_sim_score': embedding_sim_map\n}\ngenerate_raw_recall_output(\n    customers_to_predict=target_customers,\n    all_transactions=transactions,\n    brain_maps=content_brains,\n    filename=\"raw_recall_content_based_with_scores.csv\"\n)\n\n# --- FILE 3: OUTPUT THÔ KẾT HỢP ---\nall_brains = {\n    'transactional_score': transaction_map,\n    'name_sim_score': name_sim_map,\n    'embedding_sim_score': embedding_sim_map\n}\ngenerate_raw_recall_output(\n    customers_to_predict=target_customers,\n    all_transactions=transactions,\n    brain_maps=all_brains,\n    filename=\"raw_recall_combined_with_scores.csv\"\n)\n\nprint(\"\\n*** TUYỆT VỜI! Đã tạo xong 3 file output recall thô (có score) trong thư mục /kaggle/working/. ***\")\nprint(\"Các file này là đầu vào hoàn hảo cho mô hình xếp hạng (ranking) tiếp theo của bạn.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}