{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transections = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\"\narticals = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\"\nsample_submission = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\"\ncustomers = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv\"\n\ntransactions_df = pd.read_csv(transections)\narticles_df = pd.read_csv(articals)\ncustomers_df = pd.read_csv(customers)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(transactions_df.head())\nprint(articles_df.head())\nprint(customers_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Let's check some basic stats","metadata":{}},{"cell_type":"code","source":"# Convert `t_dat` to a datetime type for easy date manipulation\ntransactions_df['t_dat'] = pd.to_datetime(transactions_df['t_dat'])\n\n# Get a high-level overview of the data\nprint(\"Transactions DataFrame Info:\")\nprint(transactions_df.info())\n\n# Count the number of unique customers and articles\nnum_unique_customers = transactions_df['customer_id'].nunique()\nnum_unique_articles = transactions_df['article_id'].nunique()\n\n# Find the date range of the transactions\nstart_date = transactions_df['t_dat'].min()\nend_date = transactions_df['t_dat'].max()\n\n# Print the key statistics\nprint(\"\\nBasic Statistics\")\nprint(f\"Number of unique customers: {num_unique_customers}\")\nprint(f\"Number of unique articles: {num_unique_articles}\")\nprint(f\"Date range of transactions: {start_date.strftime('%Y-%m-%d')} to {end_date.strftime('%Y-%m-%d')}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# --- EDA on Customers Data ---\n#### To understand the dataset better, the next logical step in our EDA is to visualize how the number of transactions changes over time. This can reveal seasonal patterns, trends, and key events that influence purchasing behavior.","metadata":{}},{"cell_type":"code","source":"print(\"--- Customers Dataframe EDA ---\")\nprint(\"\\nMissing values in customers data:\")\nprint(customers_df.isnull().sum())\nprint(\"\\nDistribution of club_member_status:\")\nprint(customers_df['club_member_status'].value_counts())\nprint(\"\\nDistribution of fashion_news_frequency:\")\nprint(customers_df['fashion_news_frequency'].value_counts())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Customers Data: We have some key categorical features like club_member_status and fashion_news_frequency that can be very powerful predictors. The high number of missing values for FN and Active suggests that for many customers, this information isn't available. We'll need a strategy to handle these missing values, possibly by treating them as their own category. The age column, while mostly complete, has some missing values we can impute later.","metadata":{}},{"cell_type":"markdown","source":"# --- EDA on Articles Data ---","metadata":{}},{"cell_type":"code","source":"print(\"\\n--- Articles Dataframe EDA ---\")\nprint(\"\\nMissing values in articles data:\")\nprint(articles_df.isnull().sum())\nprint(\"\\nDistribution of product_group_name:\")\nprint(articles_df['product_group_name'].value_counts())\nprint(\"\\nDistribution of garment_group_name:\")\nprint(articles_df['garment_group_name'].value_counts())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Articles Data: This data is very clean! Nearly every column is complete, giving us a rich set of features to work with. The product_group_name and garment_group_name columns are particularly useful and show us the dominance of clothing categories like 'Garment Upper body' and 'Garment Lower body'.","metadata":{}},{"cell_type":"markdown","source":"### Let's merge 3 data frames with sample data","metadata":{}},{"cell_type":"code","source":"# Convert `t_dat` to datetime\ntransactions_df['t_dat'] = pd.to_datetime(transactions_df['t_dat'])\n\n# Filter transactions for the last 3 months\nend_date = transactions_df['t_dat'].max()\nstart_date_filtered = end_date - pd.DateOffset(months=3)\nrecent_transactions = transactions_df[transactions_df['t_dat'] >= start_date_filtered]\n\n# Merge the dataframes\nmerged_df = pd.merge(recent_transactions, customers_df, on='customer_id', how='left')\nmerged_df = pd.merge(merged_df, articles_df, on='article_id', how='left')\n\n# Display the information of the new, merged dataframe\nprint(\"Merged DataFrame Info (last 3 months):\")\nprint(merged_df.info())\n\nprint(\"\\nMerged DataFrame Head:\")\nmerged_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Let's fill the null data","metadata":{}},{"cell_type":"code","source":"# Let's fill null with the median age of the customers\nmedian_age = merged_df['age'].median()\nmerged_df['age'].fillna(median_age, inplace=True)\n\n# Impute missing categorical values with a placeholder\nmerged_df['club_member_status'] = merged_df['club_member_status'].fillna('Unknown')\nmerged_df['fashion_news_frequency'] = merged_df['fashion_news_frequency'].fillna('Unknown')\nmerged_df['FN'] = merged_df['FN'].fillna(0)\nmerged_df['Active'] = merged_df['Active'].fillna(0)\n\nprint(merged_df[['club_member_status', 'fashion_news_frequency', 'FN', 'Active']].isna().sum())\n\n# Create temporal features\nmerged_df['week'] = merged_df['t_dat'].dt.isocalendar().week.astype(int)\nmerged_df['day_of_week'] = merged_df['t_dat'].dt.dayofweek.astype(int)\n\nmerged_df[['t_dat', 'age', 'FN', 'Active', 'club_member_status', 'week', 'day_of_week']].head()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# calculate recency, frequency, and monetary value for each customer\ncustomer_features = merged_df.groupby('customer_id').agg(\n    total_purchase = ('article_id', 'count'),\n    last_purchase_date = ('t_dat', 'max')\n)\ncustomer_features[\"recency_days\"] = (merged_df['t_dat'].max() - customer_features['last_purchase_date']).dt.days\ncustomer_features.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# calculate popularity and average price for each article\nartical_features = merged_df.groupby('article_id').agg(\n    purchase_count = ('customer_id', 'count'),\n    average_price = ('price', 'mean')\n)\nartical_features.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Now it's time to prepare the data model\n1. generate positive samples\n2. generate negative samples\n3. merge features\n4. finalizing the datasets","metadata":{}},{"cell_type":"code","source":"# To make the process faster, let's work with a small sample of the merged data\nsample_merged_df = merged_df.sample(n=800000, random_state=42).reset_index(drop=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create positive samples with a label of 1\npositive_samples = sample_merged_df[['customer_id', 'article_id']].copy()\npositive_samples['label'] = 1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Negative Sampling: Get all unique article IDs\nall_article_ids = sample_merged_df['article_id'].unique()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# generate negetive sample list\nnegative_samples_list = []\nfor customer in positive_samples['customer_id'].unique():\n    customer_purchases = set(positive_samples[positive_samples['customer_id'] == customer]['article_id'])\n    \n    # Get articles not purchased by the customer\n    non_purchased_articles = np.setdiff1d(all_article_ids, list(customer_purchases))\n    \n    # Randomly sample a few non-purchased items for each purchase\n    num_neg_samples = min(len(non_purchased_articles), 4) # Take 4 negative samples for each positive\n    if num_neg_samples > 0:\n        neg_articles = np.random.choice(non_purchased_articles, num_neg_samples, replace=False)\n        for neg_article in neg_articles:\n            negative_samples_list.append([customer, neg_article, 0])\n\nnegative_samples = pd.DataFrame(negative_samples_list, columns=['customer_id', 'article_id', 'label'])\n\nnegative_samples.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now combine positive and negative samples\nfinal_data = pd.concat([positive_samples, negative_samples], ignore_index=True)\n\n# Merge our aggregated features\nfinal_data = pd.merge(final_data, customer_features, on='customer_id', how='left')\nfinal_data = pd.merge(final_data, artical_features, on='article_id', how='left')\n\n# Show the final dataset structure\nprint(\"\\nFinal Dataset Shape:\")\nprint(final_data.shape)\nprint(\"\\nDistribution of Labels:\")\nprint(final_data['label'].value_counts())\n\nprint(\"Final Dataset for Modeling Head:\")\nfinal_data.head()\n\n\n\n\n\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Now that we have our final dataset, with both positive and negative samples, our next step is to build and train a machine learning model. This is where we will use all the features we engineered to create a powerful ranking model.\nNow that we have our final dataset, with both positive and negative samples, our next step is to build and train a machine learning model. This is where we will use all the features we engineered to create a powerful ranking model.\n\n## Building the Ranking Model with LightGBM\n#### A great choice for this type of structured, tabular data is a gradient boosting model. They are fast, efficient, and often perform very well in competitions. We'll use LightGBM, a popular and high-performance library.\n\nThe process will involve these key steps:\n\n1. Splitting the Data: We will split our final dataset into a training set and a validation set.\n\n2. Defining Features: We will identify the features that our model will use to make predictions.\n\n3. Training the Model: We'll train a LightGBM classifier to predict the label (purchase or not) for each customer-article pair.","metadata":{}},{"cell_type":"markdown","source":"## We need to import sklearn and LightGBM model","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, accuracy_score\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\n\n# 特征与目标变量\nfeatures = ['total_purchase', 'recency_days', 'purchase_count', 'average_price']\ntarget = 'label'\n\n# 处理缺失值\nfinal_data.dropna(subset=features, inplace=True)\nX = final_data[features]\ny = final_data[target]\n\n# 数据集拆分\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n\n# LightGBM 模型\nlgb_model = lgb.LGBMClassifier(\n    objective='binary',\n    boosting_type='gbdt',\n    num_leaves=63,\n    learning_rate=0.05,\n    n_estimators=500,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    random_state=42,\n    device='gpu',\n    gpu_platform_id=0,\n    gpu_device_id=0,\n)\n\n# 训练 LightGBM 模型\nlgb_model.fit(X_train, y_train)\nprint(\"LightGBM (GPU) 模型训练完成!\")\nprint(\"LightGBM 准确率:\", lgb_model.score(X_val, y_val))\n\n# 判断设备（GPU / CPU）\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"使用设备:\", device)\n\n# 将数据转换为 Tensor\nX_train_t = torch.tensor(X_train.values, dtype=torch.float32).to(device)\ny_train_t = torch.tensor(y_train.values, dtype=torch.float32).view(-1, 1).to(device)\nX_val_t = torch.tensor(X_val.values, dtype=torch.float32).to(device)\ny_val_t = torch.tensor(y_val.values, dtype=torch.float32).view(-1, 1).to(device)\n\n# 创建训练数据加载器\ntrain_loader = DataLoader(TensorDataset(X_train_t, y_train_t), batch_size=512, shuffle=True)\n\n# 定义 MLP 模型\nclass MLP(nn.Module):\n    def __init__(self, input_dim):\n        super(MLP, self).__init__()\n        self.model = nn.Sequential(\n            nn.Linear(input_dim, 128),\n            nn.BatchNorm1d(128),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(128, 64),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(64, 1),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        return self.model(x)\n\n# 初始化 MLP 模型\nmlp_model = MLP(input_dim=X_train.shape[1]).to(device)\ncriterion = nn.BCELoss()\noptimizer = optim.Adam(mlp_model.parameters(), lr=1e-3)\n\n# 训练 MLP 模型\nepochs = 15\nmlp_model.train()\nfor epoch in range(epochs):\n    total_loss = 0\n    for xb, yb in train_loader:\n        optimizer.zero_grad()\n        preds = mlp_model(xb)\n        loss = criterion(preds, yb)\n        loss.backward()\n        optimizer.step()\n        total_loss += loss.item()\n    print(f\"Epoch {epoch+1}/{epochs}, Loss: {total_loss/len(train_loader):.4f}\")\n\n# MLP 模型评估\nmlp_model.eval()\nwith torch.no_grad():\n    mlp_pred_prob = mlp_model(X_val_t).cpu().numpy().flatten()\nmlp_pred = (mlp_pred_prob > 0.5).astype(int)\nprint(\"MLP (GPU) 模型训练完成!\")\n\n# LightGBM 预测\nlgb_pred_prob = lgb_model.predict_proba(X_val)[:, 1]\nlgb_pred = (lgb_pred_prob > 0.5).astype(int)\n\n# 计算 AUC 和准确率\nlgb_auc = roc_auc_score(y_val, lgb_pred_prob)\nmlp_auc = roc_auc_score(y_val.to_numpy(), mlp_pred_prob)  # 这里转换为 NumPy 数组\n\nprint(\"\\n=== 模型比较 ===\")\nprint(f\"LightGBM -> AUC: {lgb_auc:.4f}, ACC: {accuracy_score(y_val, lgb_pred):.4f}\")\nprint(f\"MLP (GPU)-> AUC: {mlp_auc:.4f}, ACC: {accuracy_score(y_val, mlp_pred):.4f}\")\n\n# 选择最好的模型\nbest_model = lgb_model if lgb_auc >= mlp_auc else mlp_model\nmodel_name = \"LightGBM (GPU)\" if lgb_auc >= mlp_auc else \"MLP (GPU)\"\nprint(f\"\\n使用 {model_name} 生成提交文件\")\n\n# 生成测试集预测结果\ntest_df = final_data.copy()\nX_test = test_df[features]\n\nif model_name.startswith(\"LightGBM\"):\n    test_pred_prob = best_model.predict_proba(X_test)[:, 1]\nelse:\n    X_test_t = torch.tensor(X_test.values, dtype=torch.float32).to(device)\n    with torch.no_grad():\n        test_pred_prob = best_model(X_test_t).cpu().numpy().flatten()\n\n# 构建提交文件\npred_df = pd.DataFrame({\n    'customer_id': test_df['customer_id'],\n    'article_id': test_df['article_id'].astype(str).str.zfill(10),\n    'pred_prob': test_pred_prob\n}).sort_values(['customer_id', 'pred_prob'], ascending=[True, False])\n\ntop12_df = (\n    pred_df.groupby('customer_id')['article_id']\n    .apply(lambda x: ' '.join(x.head(12)))\n    .reset_index()\n    .rename(columns={'article_id': 'prediction'})\n)\n\n# 读取样本提交文件\nsample_sub = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")\nsubmission = sample_sub[['customer_id']].merge(top12_df, on='customer_id', how='left')\nsubmission['prediction'] = submission['prediction'].fillna('')\nassert len(submission) == len(sample_sub)\nassert submission['customer_id'].equals(sample_sub['customer_id'])\n\n# 保存提交文件\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"\\n提交文件已生成: submission.csv\")\nprint(\"文件形状:\", submission.shape)\nprint(submission.head())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  分析特征重要性 \nimport matplotlib.pyplot as plt\nfeature_importance = (\n    pd.DataFrame({\n        'feature': features,\n        'importance': lgb_model.feature_importances_\n    })\n    .sort_values(by='importance', ascending=False)\n)\n\nprint(\"\\n 特征重要性：\")\nprint(feature_importance)\n\n# 可视化\nplt.figure(figsize=(6,4))\nplt.barh(feature_importance['feature'], feature_importance['importance'])\nplt.gca().invert_yaxis()\nplt.title('Feature Importance (LightGBM)')\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score, accuracy_score, classification_report, roc_curve\nimport matplotlib.pyplot as plt\n\n#  预测概率与分类\ny_pred_prob = lgb_model.predict_proba(X_val)[:, 1]\ny_pred = (y_pred_prob > 0.5).astype(int)\n\n#  主要评估指标\nauc = roc_auc_score(y_val, y_pred_prob)\nacc = accuracy_score(y_val, y_pred)\n\nprint(\"\\n 模型评估结果：\")\nprint(f\"AUC: {auc:.4f}\")\nprint(f\"Accuracy: {acc:.4f}\")\nprint(\"\\n分类报告:\")\nprint(classification_report(y_val, y_pred))\n\n#  ROC曲线可视化\nfpr, tpr, _ = roc_curve(y_val, y_pred_prob)\nplt.figure(figsize=(6, 5))\nplt.plot(fpr, tpr, label=f'LightGBM (AUC={auc:.4f})')\nplt.plot([0, 1], [0, 1], '--', color='gray')\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}