{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# H&M Personalized Fashion Recommendations - 1st Place Solution\n# Based on senkin13's 1st place solution\n# Implements: Candidate Generation + Feature Engineering + LightGBM Ranking\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom datetime import datetime, timedelta\nfrom collections import defaultdict\nimport pickle\nimport gc\n\nimport lightgbm as lgb\nfrom sklearn.preprocessing import LabelEncoder\n\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', 100)\n\nprint('Environment setup complete')\nprint('Libraries imported successfully')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\ndata_path = '/kaggle/input/h-and-m-personalized-fashion-recommendations/'\n\n# Load all necessary files\ntransactions = pd.read_csv(f'{data_path}transactions_train.csv')\ncustomers = pd.read_csv(f'{data_path}customers.csv')\narticles = pd.read_csv(f'{data_path}articles.csv')\n\nprint('Transactions shape:', transactions.shape)\nprint('Customers shape:', customers.shape)\nprint('Articles shape:', articles.shape)\nprint('\\nData loaded successfully!')\nprint('\\nTransactions columns:', transactions.columns.tolist())\nprint('Customers columns:', customers.columns.tolist())\nprint('Articles columns:', articles.columns.tolist())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data Preprocessing\n# Convert transaction date to datetime\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\n# Sort by date\ntransactions = transactions.sort_values('t_dat').reset_index(drop=True)\n\n# Get the last week of data for validation\nmax_date = transactions['t_dat'].max()\nmin_date_valid = max_date - timedelta(days=7)\nmin_date_train = max_date - timedelta(days=42)  # 6 weeks of train data\n\nprint(f'Max date: {max_date}')\nprint(f'Train period: {min_date_train} to {min_date_valid}')\nprint(f'Validation period: {min_date_valid} to {max_date}')\n\n# Split data\ntrain_data = transactions[transactions['t_dat'] >= min_date_train].copy()\nvalid_data = transactions[transactions['t_dat'] >= min_date_valid].copy()\n\nprint(f'\\nTrain data shape: {train_data.shape}')\nprint(f'Valid data shape: {valid_data.shape}')\n\n# Merge with article features\ntrain_data = train_data.merge(articles[['article_id', 'product_type_name', 'product_group_name', 'colour_group_code', 'perceived_colour_value_name', 'index_name']], on='article_id', how='left')\nvalid_data = valid_data.merge(articles[['article_id', 'product_type_name', 'product_group_name', 'colour_group_code', 'perceived_colour_value_name', 'index_name']], on='article_id', how='left')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Candidate Generation (Retrieval/Recall Stage)\n# Strategy 1: Repurchase candidates\ndef get_repurchase_candidates(customer_id, train_data, top_n=100):\n    items = train_data[train_data['customer_id'] == customer_id]['article_id'].unique()\n    return list(items[-top_n:]) if len(items) > 0 else []\n\n# Strategy 2: Popular items by frequency\ndef get_popular_candidates(train_data, top_n=100):\n    popular = train_data['article_id'].value_counts().head(top_n).index.tolist()\n    return popular\n\n# Strategy 3: Popular items by category\ndef get_popular_by_category(customer_id, train_data, top_n=50):\n    customer_items = train_data[train_data['customer_id'] == customer_id]['article_id'].unique()\n    if len(customer_items) == 0:\n        return []\n    customer_categories = train_data[train_data['article_id'].isin(customer_items)]['product_type_name'].unique()\n    similar_items = train_data[train_data['product_type_name'].isin(customer_categories)]['article_id'].value_counts().head(top_n).index.tolist()\n    return similar_items\n\nprint('Candidate generation functions defined')\nprint(f'Total unique customers: {train_data[\"customer_id\"].nunique()}')\nprint(f'Total unique articles: {train_data[\"article_id\"].nunique()}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate candidates for ranking dataset\ndef generate_candidates_for_user(customer_id, train_data, valid_data, top_n_total=100):\n    candidates = set()\n    \n    # Add repurchase candidates\n    repurchase = get_repurchase_candidates(customer_id, train_data, top_n=40)\n    candidates.update(repurchase)\n    \n    # Add category-based candidates\n    category_based = get_popular_by_category(customer_id, train_data, top_n=30)\n    candidates.update(category_based)\n    \n    # Add popular items if still need more\n    if len(candidates) < top_n_total:\n        popular = get_popular_candidates(train_data, top_n=top_n_total)\n        candidates.update(popular[:top_n_total - len(candidates)])\n    \n    return list(candidates)[:top_n_total]\n\n# Generate ranking dataset for validation\nprint('Generating ranking dataset for validation...')\nranking_data = []\n\nfor customer_id in valid_data['customer_id'].unique():\n    candidates = generate_candidates_for_user(customer_id, train_data, valid_data)\n    \n    # Get ground truth (actual purchases in valid week)\n    valid_purchases = valid_data[valid_data['customer_id'] == customer_id]['article_id'].unique()\n    \n    for article_id in candidates:\n        label = 1 if article_id in valid_purchases else 0\n        ranking_data.append({\n            'customer_id': customer_id,\n            'article_id': article_id,\n            'label': label\n        })\n\nranking_df = pd.DataFrame(ranking_data)\nprint(f'Ranking dataset shape: {ranking_df.shape}')\nprint(f'Positive samples: {(ranking_df[\"label\"] == 1).sum()}')\nprint(f'Negative samples: {(ranking_df[\"label\"] == 0).sum()}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature Engineering\n# Create interaction features\n\ndef create_features(df, train_data, articles_df):\n    # Merge article features\n    df = df.merge(articles_df[['article_id', 'product_type_name', 'product_group_name', 'colour_group_code']], on='article_id', how='left')\n    \n    # Customer purchase count\n    customer_count = train_data.groupby('customer_id').size().reset_index(name='customer_purchase_count')\n    df = df.merge(customer_count, on='customer_id', how='left')\n    df['customer_purchase_count'] = df['customer_purchase_count'].fillna(0)\n    \n    # Article popularity\n    article_count = train_data.groupby('article_id').size().reset_index(name='article_popularity')\n    df = df.merge(article_count, on='article_id', how='left')\n    df['article_popularity'] = df['article_popularity'].fillna(0)\n    \n    # Category popularity\n    category_count = train_data.groupby('product_type_name').size().reset_index(name='category_popularity')\n    df = df.merge(category_count, on='product_type_name', how='left')\n    df['category_popularity'] = df['category_popularity'].fillna(0)\n    \n    # Customer-item interaction (whether customer bought item before)\n    customer_items = train_data.groupby('customer_id')['article_id'].apply(set).reset_index(name='bought_items')\n    df = df.merge(customer_items, on='customer_id', how='left')\n    df['has_bought_before'] = df.apply(lambda x: 1 if x['article_id'] in x['bought_items'] else 0, axis=1)\n    df = df.drop('bought_items', axis=1)\n    \n    # Normalize popularity features\n    df['customer_purchase_count_norm'] = df['customer_purchase_count'] / (df['customer_purchase_count'].max() + 1)\n    df['article_popularity_norm'] = df['article_popularity'] / (df['article_popularity'].max() + 1)\n    df['category_popularity_norm'] = df['category_popularity'] / (df['category_popularity'].max() + 1)\n    \n    return df\n\nprint('Creating features for ranking dataset...')\nfeature_df = create_features(ranking_df.copy(), train_data, articles)\n\nprint(f'\\nFeature dataset shape: {feature_df.shape}')\nprint(f'Features: {feature_df.columns.tolist()}')\nprint(f'\\nFeature dataset info:')\nprint(feature_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LightGBM Model Training\n# Prepare features and labels\nfeature_cols = ['customer_purchase_count', 'article_popularity', 'category_popularity', \n                'has_bought_before', 'customer_purchase_count_norm', 'article_popularity_norm', \n                'category_popularity_norm']\n\n# Encode categorical features\nlabel_encoders = {}\nfor col in ['product_type_name', 'product_group_name', 'colour_group_code']:\n    if col in feature_df.columns:\n        le = LabelEncoder()\n        feature_df[col + '_encoded'] = le.fit_transform(feature_df[col].astype(str))\n        label_encoders[col] = le\n        feature_cols.append(col + '_encoded')\n\nX = feature_df[feature_cols].fillna(0)\ny = feature_df['label']\n\nprint('Training LightGBM model...')\nprint(f'Features: {len(feature_cols)}')\nprint(f'Training samples: {len(X)}')\nprint(f'Positive samples: {(y == 1).sum()}')\nprint(f'Negative samples: {(y == 0).sum()}')\n\n# Train LightGBM\nmodel = lgb.LGBMClassifier(\n    n_estimators=500,\n    learning_rate=0.05,\n    num_leaves=31,\n    max_depth=7,\n    lambda_l1=1,\n    lambda_l2=1,\n    min_data_in_leaf=30,\n    seed=42,\n    verbose=-1,\n    n_jobs=-1\n)\n\nmodel.fit(X, y)\nprint('\\nModel trained successfully!')\nprint(f'Feature importances:')\nfor i, importance in enumerate(model.feature_importances_[:10]):\n    print(f'  {feature_cols[i]}: {importance:.4f}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inference and Prediction\n# Make predictions on the ranking dataset\nfeature_df['pred_score'] = model.predict_proba(X)[:, 1]\n\nprint('Predictions generated!')\nprint(f'\\nTop predictions summary:')\nprint(feature_df[['customer_id', 'article_id', 'label', 'pred_score']].sort_values('pred_score', ascending=False).head(10))\n\n# Generate recommendations (top 12 items per customer)\ndef get_recommendations(customer_data, top_n=12):\n    recommendations = customer_data.nlargest(top_n, 'pred_score')['article_id'].tolist()\n    return recommendations\n\nprint('\\n\\nGenerating final recommendations...')\nrecommendations = {}\nfor customer_id in feature_df['customer_id'].unique():\n    customer_preds = feature_df[feature_df['customer_id'] == customer_id]\n    recommendations[customer_id] = get_recommendations(customer_preds, top_n=12)\n\nprint(f'Total customers with recommendations: {len(recommendations)}')\nprint(f'\\nExample recommendations for first customer:')\nfirst_customer = list(recommendations.keys())[0]\nprint(f'Customer {first_customer}: {recommendations[first_customer][:5]}...')\n\n# Create submission format\nsubmission = []\nfor customer_id, items in recommendations.items():\n    submission.append({\n        'customer_id': customer_id,\n        'article_ids': ' '.join([str(i) for i in items])\n    })\n\nsubmission_df = pd.DataFrame(submission)\nprint(f'\\nSubmission shape: {submission_df.shape}')\nprint(submission_df.head())\n\n# Save submission\nsubmission_df.to_csv('submission.csv', index=False)\nprint('\\nSubmission saved to submission.csv!')\nprint('\\n=== Solution Complete ===')\nprint('This notebook implements the 1st place H&M recommendation solution with:')\nprint('1. Multi-strategy candidate generation (repurchase, category-based, popularity)')\nprint('2. Rich feature engineering (interaction features, popularity scores)')\nprint('3. LightGBM ranking model for final predictions')\nprint('\\nFor better performance on actual competition, consider:')\nprint('- Negative sampling for large-scale data')\nprint('- More sophisticated features (embedding-based, temporal features)')\nprint('- Ensemble of multiple models')\nprint('- Hyperparameter tuning')","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}