{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:05.566004Z","iopub.execute_input":"2024-11-12T18:41:05.5667Z","iopub.status.idle":"2024-11-12T18:41:06.628271Z","shell.execute_reply.started":"2024-11-12T18:41:05.566655Z","shell.execute_reply":"2024-11-12T18:41:06.627486Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# Fetcing all the data available and converting them to dataframes.Extracting basic information like shape, duplicates or null values from the data.\n","metadata":{}},{"cell_type":"markdown","source":"# Articles Data","metadata":{}},{"cell_type":"code","source":"df_articles = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:06.630069Z","iopub.execute_input":"2024-11-12T18:41:06.630557Z","iopub.status.idle":"2024-11-12T18:41:07.690239Z","shell.execute_reply.started":"2024-11-12T18:41:06.630522Z","shell.execute_reply":"2024-11-12T18:41:07.689386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_articles.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:07.691313Z","iopub.execute_input":"2024-11-12T18:41:07.691597Z","iopub.status.idle":"2024-11-12T18:41:07.698559Z","shell.execute_reply.started":"2024-11-12T18:41:07.691567Z","shell.execute_reply":"2024-11-12T18:41:07.697629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_articles.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:07.701368Z","iopub.execute_input":"2024-11-12T18:41:07.701744Z","iopub.status.idle":"2024-11-12T18:41:07.733519Z","shell.execute_reply.started":"2024-11-12T18:41:07.701701Z","shell.execute_reply":"2024-11-12T18:41:07.732636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_articles.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:07.73463Z","iopub.execute_input":"2024-11-12T18:41:07.73495Z","iopub.status.idle":"2024-11-12T18:41:07.741106Z","shell.execute_reply.started":"2024-11-12T18:41:07.734918Z","shell.execute_reply":"2024-11-12T18:41:07.740149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_articles['article_id'].count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:07.742282Z","iopub.execute_input":"2024-11-12T18:41:07.742588Z","iopub.status.idle":"2024-11-12T18:41:07.752875Z","shell.execute_reply.started":"2024-11-12T18:41:07.742555Z","shell.execute_reply":"2024-11-12T18:41:07.752008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"product_dict = df_articles.set_index('product_code')['prod_name'].to_dict()\nproduct_type = df_articles.set_index('product_type_no')['product_type_name'].to_dict()\ngraphical_appearance = df_articles.set_index('graphical_appearance_no')['graphical_appearance_name'].to_dict()\ncolour_group = df_articles.set_index('colour_group_code')['colour_group_name'].to_dict()\nperceived_colour_value = df_articles.set_index('perceived_colour_value_id')['perceived_colour_value_name'].to_dict()\nperceived_colour_master = df_articles.set_index('perceived_colour_master_id')['perceived_colour_master_name'].to_dict()\ndepartment = df_articles.set_index('department_no')['department_name'].to_dict()\nindex = df_articles.set_index('index_code')['index_name'].to_dict()\nindex_group = df_articles.set_index('index_group_no')['index_group_name'].to_dict()\nsection = df_articles.set_index('section_no')['section_name'].to_dict()\ngarment_group = df_articles.set_index('garment_group_no')['garment_group_name'].to_dict()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:07.754137Z","iopub.execute_input":"2024-11-12T18:41:07.754885Z","iopub.status.idle":"2024-11-12T18:41:09.318488Z","shell.execute_reply.started":"2024-11-12T18:41:07.754822Z","shell.execute_reply":"2024-11-12T18:41:09.31743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#product_type","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:09.319669Z","iopub.execute_input":"2024-11-12T18:41:09.319966Z","iopub.status.idle":"2024-11-12T18:41:09.32422Z","shell.execute_reply.started":"2024-11-12T18:41:09.319934Z","shell.execute_reply":"2024-11-12T18:41:09.323277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_articles.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:09.325454Z","iopub.execute_input":"2024-11-12T18:41:09.325865Z","iopub.status.idle":"2024-11-12T18:41:09.509641Z","shell.execute_reply.started":"2024-11-12T18:41:09.325797Z","shell.execute_reply":"2024-11-12T18:41:09.508684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_articles.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:09.513383Z","iopub.execute_input":"2024-11-12T18:41:09.513679Z","iopub.status.idle":"2024-11-12T18:41:09.655916Z","shell.execute_reply.started":"2024-11-12T18:41:09.513648Z","shell.execute_reply":"2024-11-12T18:41:09.654915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_art = df_articles.drop(['prod_name','product_type_name','graphical_appearance_name','colour_group_name','perceived_colour_value_name','perceived_colour_master_name','department_name','index_name','index_group_name','section_name','garment_group_name'],axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:09.657181Z","iopub.execute_input":"2024-11-12T18:41:09.657511Z","iopub.status.idle":"2024-11-12T18:41:09.66726Z","shell.execute_reply.started":"2024-11-12T18:41:09.657478Z","shell.execute_reply":"2024-11-12T18:41:09.666401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_art.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:09.66851Z","iopub.execute_input":"2024-11-12T18:41:09.668924Z","iopub.status.idle":"2024-11-12T18:41:09.675369Z","shell.execute_reply.started":"2024-11-12T18:41:09.66888Z","shell.execute_reply":"2024-11-12T18:41:09.674515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:09.676603Z","iopub.execute_input":"2024-11-12T18:41:09.676915Z","iopub.status.idle":"2024-11-12T18:41:15.048984Z","shell.execute_reply.started":"2024-11-12T18:41:09.676882Z","shell.execute_reply":"2024-11-12T18:41:15.04812Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Customer Data ","metadata":{}},{"cell_type":"code","source":"df_customers.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:15.050411Z","iopub.execute_input":"2024-11-12T18:41:15.05079Z","iopub.status.idle":"2024-11-12T18:41:15.056734Z","shell.execute_reply.started":"2024-11-12T18:41:15.050745Z","shell.execute_reply":"2024-11-12T18:41:15.055889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:15.057976Z","iopub.execute_input":"2024-11-12T18:41:15.058283Z","iopub.status.idle":"2024-11-12T18:41:16.717847Z","shell.execute_reply.started":"2024-11-12T18:41:15.058252Z","shell.execute_reply":"2024-11-12T18:41:16.716893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:16.719174Z","iopub.execute_input":"2024-11-12T18:41:16.719575Z","iopub.status.idle":"2024-11-12T18:41:17.244224Z","shell.execute_reply.started":"2024-11-12T18:41:16.719529Z","shell.execute_reply":"2024-11-12T18:41:17.24329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:17.245443Z","iopub.execute_input":"2024-11-12T18:41:17.245721Z","iopub.status.idle":"2024-11-12T18:41:17.252332Z","shell.execute_reply.started":"2024-11-12T18:41:17.24569Z","shell.execute_reply":"2024-11-12T18:41:17.251317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers['Active'] = df_customers['Active'].fillna(0.0)\ndf_customers['FN'] = df_customers['FN'].fillna(0.0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:17.253583Z","iopub.execute_input":"2024-11-12T18:41:17.253901Z","iopub.status.idle":"2024-11-12T18:41:17.284257Z","shell.execute_reply.started":"2024-11-12T18:41:17.253867Z","shell.execute_reply":"2024-11-12T18:41:17.283306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:17.285492Z","iopub.execute_input":"2024-11-12T18:41:17.285794Z","iopub.status.idle":"2024-11-12T18:41:17.807248Z","shell.execute_reply.started":"2024-11-12T18:41:17.285761Z","shell.execute_reply":"2024-11-12T18:41:17.806279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers['club_member_status'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:17.808492Z","iopub.execute_input":"2024-11-12T18:41:17.808821Z","iopub.status.idle":"2024-11-12T18:41:18.023445Z","shell.execute_reply.started":"2024-11-12T18:41:17.808786Z","shell.execute_reply":"2024-11-12T18:41:18.022428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers['fashion_news_frequency'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:18.024823Z","iopub.execute_input":"2024-11-12T18:41:18.025329Z","iopub.status.idle":"2024-11-12T18:41:18.236895Z","shell.execute_reply.started":"2024-11-12T18:41:18.025282Z","shell.execute_reply":"2024-11-12T18:41:18.235866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder, LabelEncoder, StandardScaler\nlabel_enc = LabelEncoder()\ndf_customers['fashion_news_frequency'] = label_enc.fit_transform(df_customers['fashion_news_frequency'])\ndf_customers['club_member_status'] = label_enc.fit_transform(df_customers['club_member_status'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:18.238141Z","iopub.execute_input":"2024-11-12T18:41:18.238491Z","iopub.status.idle":"2024-11-12T18:41:19.918863Z","shell.execute_reply.started":"2024-11-12T18:41:18.238441Z","shell.execute_reply":"2024-11-12T18:41:19.917552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:19.920482Z","iopub.execute_input":"2024-11-12T18:41:19.920932Z","iopub.status.idle":"2024-11-12T18:41:20.203682Z","shell.execute_reply.started":"2024-11-12T18:41:19.920896Z","shell.execute_reply":"2024-11-12T18:41:20.202857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"age_bins = [0, 18, 25, 35, 100]\nage_labels = ['0-18', '18-25', '25-35', '35+']\ndf_customers['age_group'] = pd.cut(df_customers['age'], bins=age_bins, labels=age_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:20.205042Z","iopub.execute_input":"2024-11-12T18:41:20.205457Z","iopub.status.idle":"2024-11-12T18:41:20.246518Z","shell.execute_reply.started":"2024-11-12T18:41:20.205403Z","shell.execute_reply":"2024-11-12T18:41:20.245735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:20.247535Z","iopub.execute_input":"2024-11-12T18:41:20.247808Z","iopub.status.idle":"2024-11-12T18:41:20.263934Z","shell.execute_reply.started":"2024-11-12T18:41:20.247778Z","shell.execute_reply":"2024-11-12T18:41:20.262863Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Transaction Data","metadata":{}},{"cell_type":"code","source":"df_trans = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:41:20.265083Z","iopub.execute_input":"2024-11-12T18:41:20.26542Z","iopub.status.idle":"2024-11-12T18:42:50.730157Z","shell.execute_reply.started":"2024-11-12T18:41:20.265386Z","shell.execute_reply":"2024-11-12T18:42:50.729247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T18:42:50.731408Z","iopub.execute_input":"2024-11-12T18:42:50.731793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans.head(5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans.columns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans.isnull().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create Recency Feature (days since last purchase)\ndf_trans['t_dat'] = pd.to_datetime(df_trans['t_dat'])\ncurrent_date = pd.to_datetime('2024-11-12')  # Today's date for recency calculation\ndf_trans['days_since_last_purchase'] = (current_date - df_trans['t_dat']).dt.days","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trans","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission dataframe","metadata":{}},{"cell_type":"code","source":"df_sub = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sub","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sub.duplicated().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploring transaction data","metadata":{}},{"cell_type":"code","source":"df_trans.head(10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The total price spent by each customer","metadata":{}},{"cell_type":"code","source":"df_trans.groupby('customer_id').price.sum().reset_index()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findig the total number of items purchased by a single customer","metadata":{}},{"cell_type":"code","source":"grouped_data = df_trans.groupby(['customer_id','days_since_last_purchase']).agg({\n    'article_id': 'count',   # Number of articles purchased\n    'price': 'sum',           # Total spending\n    'sales_channel_id': 'nunique' # Number of unique sales channels used \n}).reset_index().sort_values(by = 'price',ascending = False)\ngrouped_data","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grouped_data = df_trans.groupby(['customer_id','days_since_last_purchase']).agg({\n    'article_id': 'count',   # Number of articles purchased\n    'price': 'sum',           # Total spending\n    'sales_channel_id': 'nunique' # Number of unique sales channels used \n}).reset_index().sort_values(by = ['days_since_last_purchase','price'],ascending = False)\ngrouped_data","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Number of times an article purchased by a customer","metadata":{}},{"cell_type":"code","source":"df_trans.groupby(['customer_id','article_id']).size().reset_index(name = 'frequency').sort_values(by = 'customer_id',ascending = False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Frequency of items purchased","metadata":{}},{"cell_type":"code","source":"df_trans.groupby(['article_id']).size().reset_index(name = 'frequency').sort_values(by = ['frequency'],ascending = False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Frequency of an item purchased by a single customer","metadata":{}},{"cell_type":"code","source":"df_trans.groupby(['customer_id','article_id']).size().reset_index(name = 'frequency').sort_values(by = ['frequency'],ascending = False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Total spend, purchase frequency, and average transaction value per customer.","metadata":{}},{"cell_type":"markdown","source":"Most Recent purchase of the customer","metadata":{}},{"cell_type":"markdown","source":"Most frequent item purchased by the customer","metadata":{}},{"cell_type":"markdown","source":"Ranking onn items based on the user preference","metadata":{}},{"cell_type":"code","source":"# Aggregate Transaction Features per Customer\ncustomer_transaction_features = df_trans.groupby('customer_id').agg(\n    total_spend=('price', 'sum'),\n    avg_transaction_value=('price', 'mean'),\n    transaction_count=('article_id', 'count')\n).reset_index()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_customers = pd.merge(df_customers, customer_transaction_features, on='customer_id', how='left')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Final DataFrame Ready for Model ---\nfinal_df = pd.merge(df_trans, df_customers, on='customer_id', how='left')\nfinal_df = pd.merge(final_df, df_articles, on='article_id', how='left')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nfinal_df[['age', 'total_spend', 'avg_transaction_value', 'transaction_count', 'price']] = scaler.fit_transform(\n    final_df[['age', 'total_spend', 'avg_transaction_value', 'transaction_count', 'price']]\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.columns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.head(5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_final.groupby(['customer_id,article_id'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_final = df_final.head(1000000)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Outlier Detection","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Visualize outliers in the 'total_spend', 'avg_transaction_value', and 'price' columns\noutlier_columns = ['total_spend', 'avg_transaction_value', 'price']\n\nplt.figure(figsize=(12, 6))\nfor idx, col in enumerate(outlier_columns, 1):\n    plt.subplot(1, len(outlier_columns), idx)\n    sns.boxplot(x=final_df[col])\n    plt.title(f'Box Plot of {col}')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import zscore\n\n# Calculate Z-scores for numerical features\nnumerical_features = ['total_spend', 'avg_transaction_value', 'price']\nfinal_df_zscore = final_df[numerical_features].apply(zscore)\n\n# Flagging outliers (Z-score > 3 or Z-score < -3)\noutliers_zscore = (final_df_zscore.abs() > 3).sum(axis=1)\noutliers_df = final_df[outliers_zscore > 0]  # Outliers for any of the columns\n\nprint(f\"Number of outliers detected: {len(outliers_df)}\")\nprint(outliers_df.head())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate Q1 (25th percentile) and Q3 (75th percentile) for each numerical column\nQ1 = final_df[numerical_features].quantile(0.25)\nQ3 = final_df[numerical_features].quantile(0.75)\n\n# Calculate IQR\nIQR = Q3 - Q1\n\n# Detect outliers\noutlier_condition = ((final_df[numerical_features] < (Q1 - 1.5 * IQR)) | (  final_df[numerical_features] > (Q3 + 1.5 * IQR)))\noutliers_iqr = final_df[outlier_condition.any(axis=1)]\n\nprint(f\"Number of outliers detected using IQR: {len(outliers_iqr)}\")\nprint(outliers_iqr.head())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualizing outliers detected by IQR method\nplt.figure(figsize=(10, 6))\nsns.scatterplot(data=final_df, x='age', y='total_spend', hue=outlier_condition.any(axis=1), palette={True: 'red', False: 'blue'})\nplt.title('Outliers Detected Based on IQR')\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df_no_outliers = final_df[~outlier_condition.any(axis=1)]\nprint(f\"Data shape after removing outliers: {final_df_no_outliers.shape}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clip values above or below certain thresholds for columns\nfinal_df['total_spend'] = final_df['total_spend'].clip(lower=0, upper=final_df['total_spend'].quantile(0.95))\nfinal_df['avg_transaction_value'] = final_df['avg_transaction_value'].clip(lower=0, upper=final_df['avg_transaction_value'].quantile(0.95))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace outliers with median\nfinal_df['total_spend'] = final_df['total_spend'].where(~outlier_condition['total_spend'], final_df['total_spend'].median())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply log transformation (helps to reduce the skewness of the data)\nfinal_df['total_spend_log'] = final_df['total_spend'].apply(lambda x: np.log(x + 1))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install scikit-surprise\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from surprise import Reader, Dataset, SVD\nfrom surprise.model_selection import train_test_split\nfrom surprise import accuracy\n\n# Prepare data for Collaborative Filtering\n# Use customer_id and article_id as the user-item interaction matrix\n# Create a dataframe with 'customer_id', 'article_id', and 'total_spend'\ninteraction_data = final_df[['customer_id', 'article_id', 'total_spend']]\n\n# Define the rating scale (we can use total_spend as a proxy for \"ratings\")\nreader = Reader(rating_scale=(0, interaction_data['total_spend'].max()))\n\n# Load the dataset\ndata = Dataset.load_from_df(interaction_data[['customer_id', 'article_id', 'total_spend']], reader)\n\n# Split data into training and testing sets\ntrainset, testset = train_test_split(data, test_size=0.2)\n\n# Initialize the SVD model (Singular Value Decomposition)\nmodel = SVD()\n\n# Train the model\nmodel.fit(trainset)\n\n# Generate top-k predictions for each user in the test set\nk = 10  # Set the number of top recommendations\nuser_predictions = defaultdict(list)\n\nfor uid, iid, true_r in testset:\n    pred = model.predict(uid, iid).est\n    user_predictions[uid].append((iid, pred))\n\n# Sort and keep top-k predictions for each user\ntop_k_preds = {uid: sorted(items, key=lambda x: x[1], reverse=True)[:k] for uid, items in user_predictions.items()}\n\n# Calculate Mean Average Precision (MAP)\ndef mean_average_precision(y_true, y_pred):\n    average_precisions = []\n    \n    for true_labels, pred_items in zip(y_true, y_pred):\n        # Binarize y_true to match shape of predictions (1 if article is relevant, 0 otherwise)\n        y_true_binary = [1 if item in true_labels else 0 for item in pred_items]\n        \n        # Compute Average Precision for each user\n        if np.sum(y_true_binary) > 0:  # Only consider users with relevant items\n            ap = average_precision_score(y_true_binary, list(range(len(y_true_binary), 0, -1)))\n            average_precisions.append(ap)\n    \n    return np.mean(average_precisions)\n\n# Prepare data for MAP calculation\nall_y_true = []  # True items bought by each user in testset\nall_y_pred = []  # Predicted items for each user\n\n# Get true items and predicted items for each user\nfor uid in top_k_preds.keys():\n    # True items are those that were actually bought by the user in the test set\n    true_items = [iid for (uid_, iid, r) in testset if uid_ == uid and r > 0]\n    all_y_true.append(true_items)\n    \n    # Predicted items for the user from top_k_preds\n    pred_items = [iid for (iid, _) in top_k_preds[uid]]\n    all_y_pred.append(pred_items)\n\n# Calculate MAP\nmap_score = mean_average_precision(all_y_true, all_y_pred)\nprint(f\"Mean Average Precision (MAP): {map_score}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity\n\n# Combine relevant textual features to form a description of the item\nfinal_df['item_description'] = final_df['prod_name'] + ' ' + final_df['product_type_name'] + ' ' + final_df['colour_group_name']\n\n# Apply TF-IDF Vectorization\ntfidf = TfidfVectorizer(stop_words='english')\ntfidf_matrix = tfidf.fit_transform(final_df['item_description'])\n\n# Compute cosine similarity matrix\ncosine_sim = cosine_similarity(tfidf_matrix, tfidf_matrix)\n\n# Function to get top N similar items for a given product\ndef get_recommendations(article_id, cosine_sim=cosine_sim, top_n=5):\n    idx = final_df[final_df['article_id'] == article_id].index[0]\n    sim_scores = list(enumerate(cosine_sim[idx]))\n    sim_scores = sorted(sim_scores, key=lambda x: x[1], reverse=True)\n    sim_scores = sim_scores[1:top_n+1]\n    article_indices = [i[0] for i in sim_scores]\n    return final_df.iloc[article_indices]\n\n# Calculate Mean Average Precision (MAP)\ndef calculate_map(test_set, top_n=5):\n    all_y_true = []\n    all_y_pred = []\n\n    for customer_id in test_set['customer_id'].unique():\n        # Get articles actually purchased by the customer in test set (ground truth)\n        actual_articles = test_set[test_set['customer_id'] == customer_id]['article_id'].tolist()\n        \n        # Generate recommendations for each article in the actual list and keep the top-N unique recommendations\n        recommended_articles = []\n        for article_id in actual_articles:\n            recommendations = get_recommendations(article_id, top_n=top_n)\n            recommended_articles.extend(recommendations['article_id'].tolist())\n        \n        # Keep only unique recommendations (to avoid duplicates) and limit to top-N\n        recommended_articles = list(dict.fromkeys(recommended_articles))[:top_n]\n\n        # Binarize true labels (1 if relevant, 0 otherwise)\n        y_true = [1 if article in actual_articles else 0 for article in recommended_articles]\n        \n        # Append ground truth and predictions for MAP calculation\n        all_y_true.append(y_true)\n        all_y_pred.append(list(range(len(y_true), 0, -1)))  # ranking order for AP calculation\n\n    # Compute MAP\n    map_score = np.mean([average_precision_score(y_true, y_pred) for y_true, y_pred in zip(all_y_true, all_y_pred) if sum(y_true) > 0])\n    return map_score","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_set = df_sub.sample(500)  # Replace with your test data as needed\n\n# Calculate MAP for content-based recommendations\nmap_score = calculate_map(test_set, top_n=5)\nprint(f\"Mean Average Precision (MAP) for Content-Based Recommendations: {map_score}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\n\n# Assuming df_trans is your transaction dataset\n\n# Step 1: Data Preprocessing\n\n# Convert 't_dat' to datetime\ndf_trans['t_dat'] = pd.to_datetime(df_trans['t_dat'])\ndf_trans['days_since_last_purchase'] = (df_trans['t_dat'].max() - df_trans['t_dat']).dt.days\n\n# Sort by customer_id and t_dat\ndf_trans = df_trans.sort_values(by=['customer_id', 't_dat'])\n\n# Encode article IDs as integers\narticle_encoder = LabelEncoder()\ndf_trans['article_id_encoded'] = article_encoder.fit_transform(df_trans['article_id'])\n\n# Step 2: Create sequences for each customer\n\n# Generate sequences of previous purchases (e.g., last 10 purchases) for each customer\nseq_length = 10  # This is a hyperparameter you can tune\ncustomer_sequences = []\ncustomer_labels = []\n\nfor customer_id in df_trans['customer_id'].unique():\n    customer_data = df_trans[df_trans['customer_id'] == customer_id]\n    \n    for i in range(seq_length, len(customer_data)):\n        # Sequence of previous purchases (article_ids)\n        sequence = customer_data.iloc[i-seq_length:i]['article_id_encoded'].values\n        customer_sequences.append(sequence)\n        \n        # Predict the next article purchased\n        next_article = customer_data.iloc[i]['article_id_encoded']\n        customer_labels.append(next_article)\n\n# Convert sequences to numpy arrays\nX = np.array(customer_sequences)\ny = np.array(customer_labels)\n\n# Step 3: Prepare Data for LSTM\n\n# One-hot encode the labels (since we're predicting categorical outcomes)\ny_encoded = to_categorical(y, num_classes=len(article_encoder.classes_))\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y_encoded, test_size=0.2, random_state=42)\n\n# Step 4: Build LSTM Model\n\nmodel = Sequential()\n\n# Add LSTM layer\nmodel.add(LSTM(units=64, activation='relu', input_shape=(X_train.shape[1], 1), return_sequences=False))\nmodel.add(Dropout(0.2))\n\n# Output layer (for predicting the next article)\nmodel.add(Dense(len(article_encoder.classes_), activation='softmax'))\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Step 5: Train the Model\nmodel.fit(X_train, y_train, epochs=5, batch_size=64, validation_data=(X_test, y_test))\n\n# Step 6: Predict Future Purchases\n# For a new customer, you would use the most recent sequence of purchases to predict their next purchase\ncustomer_id = 123  # Example customer ID\nlast_purchase_seq = df_trans[df_trans['customer_id'] == customer_id].tail(seq_length)['article_id_encoded'].values\nlast_purchase_seq = last_purchase_seq.reshape((1, seq_length, 1))  # Reshape for LSTM input\n\n# Predict the next article\npredicted_article = model.predict(last_purchase_seq)\npredicted_article_id = article_encoder.inverse_transform([np.argmax(predicted_article)])\n\nprint(f\"Predicted next purchase for customer {customer_id}: Article ID {predicted_article_id[0]}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import average_precision_score\n\n# Function to calculate Mean Average Precision\ndef mean_average_precision(y_true, y_pred):\n    \"\"\"\n    Calculate Mean Average Precision (MAP) at a batch level.\n\n    Args:\n        y_true: List of true labels for each customer (actual purchased articles).\n        y_pred: List of predicted labels for each customer (predicted articles, sorted by relevance).\n\n    Returns:\n        Mean Average Precision (MAP) score.\n    \"\"\"\n    average_precisions = []\n    for true_labels, pred_scores in zip(y_true, y_pred):\n        # Binarize y_true to match shape of predictions (1 for relevant items, 0 otherwise)\n        y_true_binary = [1 if item in true_labels else 0 for item in pred_scores]\n        \n        # Calculate average precision for the current customer's predictions\n        ap = average_precision_score(y_true_binary, [score for _, score in pred_scores])\n        average_precisions.append(ap)\n    \n    # Compute mean of average precision scores\n    return np.mean(average_precisions)\n\n# After training, generate predictions\n\n# Example data (replace with model predictions)\nall_y_true = []  # Each element is a list of actual articles purchased by the customer\nall_y_pred = []  # Each element is a list of (article, probability) pairs, sorted by score in descending order\n\nfor customer_id in unique_customers:  # Loop through each customer\n    last_purchase_seq = df_trans[df_trans['customer_id'] == customer_id].tail(seq_length)['article_id_encoded'].values\n    last_purchase_seq = last_purchase_seq.reshape((1, seq_length, 1))\n    \n    # Predict probabilities for all articles\n    predicted_probs = model.predict(last_purchase_seq)[0]\n    \n    # Sort articles by predicted probability\n    top_articles = sorted(list(enumerate(predicted_probs)), key=lambda x: x[1], reverse=True)\n    top_articles = [(article_encoder.inverse_transform([idx])[0], prob) for idx, prob in top_articles]\n    \n    # Store the true articles and predicted articles for MAP calculation\n    true_articles = df_trans[(df_trans['customer_id'] == customer_id) & \n                             (df_trans['t_dat'] > end_of_training_period)]['article_id'].values\n    all_y_true.append(true_articles)\n    all_y_pred.append(top_articles)\n\n# Calculate MAP for all customers\nmap_score = mean_average_precision(all_y_true, all_y_pred)\nprint(f\"Mean Average Precision (MAP) Score: {map_score}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}