{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:13:45.531285Z","iopub.execute_input":"2025-08-21T13:13:45.531590Z","iopub.status.idle":"2025-08-21T13:13:45.906936Z","shell.execute_reply.started":"2025-08-21T13:13:45.531569Z","shell.execute_reply":"2025-08-21T13:13:45.905835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transections = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\"\narticals = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\"\nsample_submission = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\"\ncustomers = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv\"\n\ntransactions_df = pd.read_csv(transections)\narticles_df = pd.read_csv(articals)\ncustomers_df = pd.read_csv(customers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:13:48.090078Z","iopub.execute_input":"2025-08-21T13:13:48.090712Z","iopub.status.idle":"2025-08-21T13:15:26.928449Z","shell.execute_reply.started":"2025-08-21T13:13:48.090680Z","shell.execute_reply":"2025-08-21T13:15:26.927486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(transactions_df.head())\nprint(articles_df.head())\nprint(customers_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:19:56.833580Z","iopub.execute_input":"2025-08-21T13:19:56.833951Z","iopub.status.idle":"2025-08-21T13:19:56.864297Z","shell.execute_reply.started":"2025-08-21T13:19:56.833924Z","shell.execute_reply":"2025-08-21T13:19:56.863074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Let's check some basic stats","metadata":{}},{"cell_type":"code","source":"# Convert `t_dat` to a datetime type for easy date manipulation\ntransactions_df['t_dat'] = pd.to_datetime(transactions_df['t_dat'])\n\n# Get a high-level overview of the data\nprint(\"Transactions DataFrame Info:\")\nprint(transactions_df.info())\n\n# Count the number of unique customers and articles\nnum_unique_customers = transactions_df['customer_id'].nunique()\nnum_unique_articles = transactions_df['article_id'].nunique()\n\n# Find the date range of the transactions\nstart_date = transactions_df['t_dat'].min()\nend_date = transactions_df['t_dat'].max()\n\n# Print the key statistics\nprint(\"\\nBasic Statistics\")\nprint(f\"Number of unique customers: {num_unique_customers}\")\nprint(f\"Number of unique articles: {num_unique_articles}\")\nprint(f\"Date range of transactions: {start_date.strftime('%Y-%m-%d')} to {end_date.strftime('%Y-%m-%d')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:21:17.500684Z","iopub.execute_input":"2025-08-21T13:21:17.503597Z","iopub.status.idle":"2025-08-21T13:21:31.066236Z","shell.execute_reply.started":"2025-08-21T13:21:17.503510Z","shell.execute_reply":"2025-08-21T13:21:31.065086Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# --- EDA on Customers Data ---\n#### To understand the dataset better, the next logical step in our EDA is to visualize how the number of transactions changes over time. This can reveal seasonal patterns, trends, and key events that influence purchasing behavior.","metadata":{}},{"cell_type":"code","source":"print(\"--- Customers Dataframe EDA ---\")\nprint(\"\\nMissing values in customers data:\")\nprint(customers_df.isnull().sum())\nprint(\"\\nDistribution of club_member_status:\")\nprint(customers_df['club_member_status'].value_counts())\nprint(\"\\nDistribution of fashion_news_frequency:\")\nprint(customers_df['fashion_news_frequency'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:25:17.054522Z","iopub.execute_input":"2025-08-21T13:25:17.054939Z","iopub.status.idle":"2025-08-21T13:25:17.733654Z","shell.execute_reply.started":"2025-08-21T13:25:17.054913Z","shell.execute_reply":"2025-08-21T13:25:17.732490Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Customers Data: We have some key categorical features like club_member_status and fashion_news_frequency that can be very powerful predictors. The high number of missing values for FN and Active suggests that for many customers, this information isn't available. We'll need a strategy to handle these missing values, possibly by treating them as their own category. The age column, while mostly complete, has some missing values we can impute later.","metadata":{}},{"cell_type":"markdown","source":"# --- EDA on Articles Data ---","metadata":{}},{"cell_type":"code","source":"print(\"\\n--- Articles Dataframe EDA ---\")\nprint(\"\\nMissing values in articles data:\")\nprint(articles_df.isnull().sum())\nprint(\"\\nDistribution of product_group_name:\")\nprint(articles_df['product_group_name'].value_counts())\nprint(\"\\nDistribution of garment_group_name:\")\nprint(articles_df['garment_group_name'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:25:46.133022Z","iopub.execute_input":"2025-08-21T13:25:46.133430Z","iopub.status.idle":"2025-08-21T13:25:46.252824Z","shell.execute_reply.started":"2025-08-21T13:25:46.133389Z","shell.execute_reply":"2025-08-21T13:25:46.251696Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Articles Data: This data is very clean! Nearly every column is complete, giving us a rich set of features to work with. The product_group_name and garment_group_name columns are particularly useful and show us the dominance of clothing categories like 'Garment Upper body' and 'Garment Lower body'.","metadata":{}},{"cell_type":"markdown","source":"### Let's merge 3 data frames with sample data","metadata":{}},{"cell_type":"code","source":"# Convert `t_dat` to datetime\ntransactions_df['t_dat'] = pd.to_datetime(transactions_df['t_dat'])\n\n# Filter transactions for the last 3 months\nend_date = transactions_df['t_dat'].max()\nstart_date_filtered = end_date - pd.DateOffset(months=3)\nrecent_transactions = transactions_df[transactions_df['t_dat'] >= start_date_filtered]\n\n# Merge the dataframes\nmerged_df = pd.merge(recent_transactions, customers_df, on='customer_id', how='left')\nmerged_df = pd.merge(merged_df, articles_df, on='article_id', how='left')\n\n# Display the information of the new, merged dataframe\nprint(\"Merged DataFrame Info (last 3 months):\")\nprint(merged_df.info())\n\nprint(\"\\nMerged DataFrame Head:\")\nmerged_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:29:12.182455Z","iopub.execute_input":"2025-08-21T13:29:12.183195Z","iopub.status.idle":"2025-08-21T13:29:21.651375Z","shell.execute_reply.started":"2025-08-21T13:29:12.183167Z","shell.execute_reply":"2025-08-21T13:29:21.650231Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Let's fill the null data","metadata":{}},{"cell_type":"code","source":"# Let's fill null with the median age of the customers\nmedian_age = merged_df['age'].median()\nmerged_df['age'].fillna(median_age, inplace=True)\n\n# Impute missing categorical values with a placeholder\nmerged_df['club_member_status'] = merged_df['club_member_status'].fillna('Unknown')\nmerged_df['fashion_news_frequency'] = merged_df['fashion_news_frequency'].fillna('Unknown')\nmerged_df['FN'] = merged_df['FN'].fillna(0)\nmerged_df['Active'] = merged_df['Active'].fillna(0)\n\nprint(merged_df[['club_member_status', 'fashion_news_frequency', 'FN', 'Active']].isna().sum())\n\n# Create temporal features\nmerged_df['week'] = merged_df['t_dat'].dt.isocalendar().week.astype(int)\nmerged_df['day_of_week'] = merged_df['t_dat'].dt.dayofweek.astype(int)\n\nmerged_df[['t_dat', 'age', 'FN', 'Active', 'club_member_status', 'week', 'day_of_week']].head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:37:21.536689Z","iopub.execute_input":"2025-08-21T13:37:21.537090Z","iopub.status.idle":"2025-08-21T13:37:23.379164Z","shell.execute_reply.started":"2025-08-21T13:37:21.537064Z","shell.execute_reply":"2025-08-21T13:37:23.378244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# calculate recency, frequency, and monetary value for each customer\ncustomer_features = merged_df.groupby('customer_id').agg(\n    total_purchase = ('article_id', 'count'),\n    last_purchase_date = ('t_dat', 'max')\n)\ncustomer_features[\"recency_days\"] = (merged_df['t_dat'].max() - customer_features['last_purchase_date']).dt.days\ncustomer_features.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:39:57.951771Z","iopub.execute_input":"2025-08-21T13:39:57.953112Z","iopub.status.idle":"2025-08-21T13:40:00.038909Z","shell.execute_reply.started":"2025-08-21T13:39:57.953072Z","shell.execute_reply":"2025-08-21T13:40:00.037500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# calculate popularity and average price for each article\nartical_features = merged_df.groupby('article_id').agg(\n    purchase_count = ('customer_id', 'count'),\n    average_price = ('price', 'mean')\n)\nartical_features.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:40:12.482092Z","iopub.execute_input":"2025-08-21T13:40:12.482533Z","iopub.status.idle":"2025-08-21T13:40:12.951300Z","shell.execute_reply.started":"2025-08-21T13:40:12.482501Z","shell.execute_reply":"2025-08-21T13:40:12.950114Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Now it's time to prepare the data model\n1. generate positive samples\n2. generate negative samples\n3. merge features\n4. finalizing the datasets","metadata":{}},{"cell_type":"code","source":"# To make the process faster, let's work with a small sample of the merged data\nsample_merged_df = merged_df.sample(n=100000, random_state=42).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:41:23.315528Z","iopub.execute_input":"2025-08-21T13:41:23.315938Z","iopub.status.idle":"2025-08-21T13:41:23.972282Z","shell.execute_reply.started":"2025-08-21T13:41:23.315909Z","shell.execute_reply":"2025-08-21T13:41:23.971206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create positive samples with a label of 1\npositive_samples = sample_merged_df[['customer_id', 'article_id']].copy()\npositive_samples['label'] = 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:41:36.183915Z","iopub.execute_input":"2025-08-21T13:41:36.184789Z","iopub.status.idle":"2025-08-21T13:41:36.208211Z","shell.execute_reply.started":"2025-08-21T13:41:36.184758Z","shell.execute_reply":"2025-08-21T13:41:36.207219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Negative Sampling: Get all unique article IDs\nall_article_ids = sample_merged_df['article_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:41:46.403769Z","iopub.execute_input":"2025-08-21T13:41:46.405087Z","iopub.status.idle":"2025-08-21T13:41:46.413494Z","shell.execute_reply.started":"2025-08-21T13:41:46.405028Z","shell.execute_reply":"2025-08-21T13:41:46.412394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# generate negetive sample list\nnegative_samples_list = []\nfor customer in positive_samples['customer_id'].unique():\n    customer_purchases = set(positive_samples[positive_samples['customer_id'] == customer]['article_id'])\n    \n    # Get articles not purchased by the customer\n    non_purchased_articles = np.setdiff1d(all_article_ids, list(customer_purchases))\n    \n    # Randomly sample a few non-purchased items for each purchase\n    num_neg_samples = min(len(non_purchased_articles), 4) # Take 4 negative samples for each positive\n    if num_neg_samples > 0:\n        neg_articles = np.random.choice(non_purchased_articles, num_neg_samples, replace=False)\n        for neg_article in neg_articles:\n            negative_samples_list.append([customer, neg_article, 0])\n\nnegative_samples = pd.DataFrame(negative_samples_list, columns=['customer_id', 'article_id', 'label'])\n\nnegative_samples.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T13:41:56.598705Z","iopub.execute_input":"2025-08-21T13:41:56.599122Z","iopub.status.idle":"2025-08-21T14:14:02.427728Z","shell.execute_reply.started":"2025-08-21T13:41:56.599090Z","shell.execute_reply":"2025-08-21T14:14:02.426725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now combine positive and negative samples\nfinal_data = pd.concat([positive_samples, negative_samples], ignore_index=True)\n\n# Merge our aggregated features\nfinal_data = pd.merge(final_data, customer_features, on='customer_id', how='left')\nfinal_data = pd.merge(final_data, artical_features, on='article_id', how='left')\n\n# Show the final dataset structure\nprint(\"\\nFinal Dataset Shape:\")\nprint(final_data.shape)\nprint(\"\\nDistribution of Labels:\")\nprint(final_data['label'].value_counts())\n\nprint(\"Final Dataset for Modeling Head:\")\nfinal_data.head()\n\n\n\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T14:21:58.883458Z","iopub.execute_input":"2025-08-21T14:21:58.883840Z","iopub.status.idle":"2025-08-21T14:21:58.904349Z","shell.execute_reply.started":"2025-08-21T14:21:58.883816Z","shell.execute_reply":"2025-08-21T14:21:58.902905Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Now that we have our final dataset, with both positive and negative samples, our next step is to build and train a machine learning model. This is where we will use all the features we engineered to create a powerful ranking model.\nNow that we have our final dataset, with both positive and negative samples, our next step is to build and train a machine learning model. This is where we will use all the features we engineered to create a powerful ranking model.\n\n## Building the Ranking Model with LightGBM\n#### A great choice for this type of structured, tabular data is a gradient boosting model. They are fast, efficient, and often perform very well in competitions. We'll use LightGBM, a popular and high-performance library.\n\nThe process will involve these key steps:\n\n1. Splitting the Data: We will split our final dataset into a training set and a validation set.\n\n2. Defining Features: We will identify the features that our model will use to make predictions.\n\n3. Training the Model: We'll train a LightGBM classifier to predict the label (purchase or not) for each customer-article pair.","metadata":{}},{"cell_type":"markdown","source":"## We need to import sklearn and LightGBM model","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T14:31:14.691888Z","iopub.execute_input":"2025-08-21T14:31:14.692295Z","iopub.status.idle":"2025-08-21T14:31:14.698520Z","shell.execute_reply.started":"2025-08-21T14:31:14.692269Z","shell.execute_reply":"2025-08-21T14:31:14.697409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define features and target\nfeatures = ['total_purchase', 'recency_days', 'purchase_count', 'average_price']\ntarget = 'label'\n\n# Drop any NaN values that might have been created during merging the datasets\nfinal_data.dropna(subset=features, inplace=True)\n\nX = final_data[features]\ny = final_data[target]\n\n# Split data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\n# Initialize and train the LightGBM model\nlgb_model = lgb.LGBMClassifier(random_state=42)\nlgb_model.fit(X_train, y_train)\n\n# Print a confirmation message\nprint(\"LightGBM model training complete!\")\nprint(\"Model accuracy on validation set:\", lgb_model.score(X_val, y_val))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T14:31:17.845460Z","iopub.execute_input":"2025-08-21T14:31:17.845773Z","iopub.status.idle":"2025-08-21T14:31:19.948554Z","shell.execute_reply.started":"2025-08-21T14:31:17.845752Z","shell.execute_reply":"2025-08-21T14:31:19.947587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}