{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"As mentioned in this [thread](https://www.kaggle.com/c/h-and-m-personalized-fashion-recommendations/discussion/307288), one way to approach this problem is to generate candidates with different models and then rank them using item features and user features. This notebook provides basic `user features` that you can use using ranking models, for example [this](https://www.kaggle.com/c/h-and-m-personalized-fashion-recommendations/discussion/309220).\n\nI am also planning to add notebook for item features and functions that combine these features and output final dataframe that can be used by model.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom abc import ABC, abstractmethod\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport pickle\nfrom collections import defaultdict\nfrom typing import List, Dict, Any, Union","metadata":{"execution":{"iopub.status.busy":"2022-02-25T17:17:16.356450Z","iopub.execute_input":"2022-02-25T17:17:16.356851Z","iopub.status.idle":"2022-02-25T17:17:16.362778Z","shell.execute_reply.started":"2022-02-25T17:17:16.356815Z","shell.execute_reply":"2022-02-25T17:17:16.362056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = Path('../input/h-and-m-personalized-fashion-recommendations')\ntransactions_train = pd.read_csv(data_path/'transactions_train.csv')\ntransactions_train['t_dat'] = pd.to_datetime(transactions_train['t_dat'])\ncustomers_df = pd.read_csv(data_path/'customers.csv')\narticles_df = pd.read_csv(data_path/'articles.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-25T17:17:19.575951Z","iopub.execute_input":"2022-02-25T17:17:19.576809Z","iopub.status.idle":"2022-02-25T17:18:39.461349Z","shell.execute_reply.started":"2022-02-25T17:17:19.576760Z","shell.execute_reply":"2022-02-25T17:18:39.460312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Basically, I am using abstraction like this below. Every class should have `get` method and should output pandas DataFrame. Then, collect all features using another class `UserFeaturesCollector`.","metadata":{}},{"cell_type":"code","source":"class UserFeatures(ABC):\n    @abstractmethod\n    def get(self) -> pd.DataFrame:\n        \"\"\"\n        customer_id -> features\n        \"\"\"\n        pass","metadata":{"execution":{"iopub.status.busy":"2022-02-25T18:26:50.331457Z","iopub.execute_input":"2022-02-25T18:26:50.332265Z","iopub.status.idle":"2022-02-25T18:26:50.336349Z","shell.execute_reply.started":"2022-02-25T18:26:50.332195Z","shell.execute_reply":"2022-02-25T18:26:50.335549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AggrFeatures(UserFeatures):\n    \"\"\"\n    basic aggregation features(min, max, mean and etc...)\n    \"\"\"\n    def __init__(self, transactions_df):\n        self.groupby_df = transactions_df.groupby('customer_id', as_index = False)\n\n    def get(self):\n        output_df = (\n            self.groupby_df['price']\n            .agg({\n                'mean_transactions': 'mean',\n                'max_transactions': 'max',\n                'min_transactions': 'min',\n                'median_transactions': 'median',\n                'sum_transactions': 'sum',\n                'max_minus_min_transactions': lambda x: x.max()-x.min()\n            })\n            .set_index('customer_id')\n            .astype('float32')\n        )\n        return output_df","metadata":{"execution":{"iopub.status.busy":"2022-02-25T18:27:37.033833Z","iopub.execute_input":"2022-02-25T18:27:37.034102Z","iopub.status.idle":"2022-02-25T18:27:37.040236Z","shell.execute_reply.started":"2022-02-25T18:27:37.034072Z","shell.execute_reply":"2022-02-25T18:27:37.039527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CountFeatures(UserFeatures):\n    \"\"\"\n    basic features connected with transactions\n    \"\"\"\n    def __init__(self, transactions_df, topk = 10):\n        self.transactions_df = transactions_df\n        self.topk = topk\n\n    def get(self):\n        grouped = self.transactions_df.groupby('customer_id', as_index = False)\n        #number of transactions, number of online articles,\n        #number of transactions bigger than mean price of transactions\n        a = (\n            grouped\n            .agg({\n                'article_id': 'count',\n                'price': lambda x: sum(np.array(x) > x.mean()),\n                'sales_channel_id': lambda x: sum(x == 2),\n            })\n            .rename(columns = {\n                'article_id': 'n_transactions',\n                'price': 'n_transactions_bigger_mean',\n                'sales_channel_id': 'n_online_articles'\n            })\n            .set_index('customer_id')\n            .astype('int8')\n        )\n        #number of unique articles, number of store articles\n        b = (\n            grouped\n            .agg({\n                'article_id': 'nunique',\n                'sales_channel_id': lambda x: sum(x == 1),\n            })\n            .rename(columns = {\n                'article_id': 'n_unique_articles',\n                'sales_channel_id': 'n_store_articles',\n            })\n            .set_index('customer_id')\n            .astype('int8')\n        )\n        #number of transactions that are in top\n        topk_articles = self.transactions_df['article_id'].value_counts()[:self.topk].index\n        c = (\n            grouped['article_id']\n            .agg({\n               f'top_article_{i}':  lambda x: sum(x == k) for i, k in enumerate(topk_articles)\n            }\n            )\n            .set_index('customer_id')\n            .astype('int8')\n        )\n        \n        output_df = a.merge(b, on = ('customer_id')).merge(c, on = ('customer_id'))\n        return output_df","metadata":{"execution":{"iopub.status.busy":"2022-02-25T18:31:55.285922Z","iopub.execute_input":"2022-02-25T18:31:55.286611Z","iopub.status.idle":"2022-02-25T18:31:55.298121Z","shell.execute_reply.started":"2022-02-25T18:31:55.286574Z","shell.execute_reply":"2022-02-25T18:31:55.297436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomerFeatures(UserFeatures):\n    \"\"\"\n    All columns from customers dataframe\n    \"\"\"\n    def __init__(self, customers_df):\n        self.customers_df = self._prepare_customers(customers_df)\n    \n    def _prepare_customers(self, customers_df):\n        customers_df['FN'] = customers_df['FN'].fillna(0).astype('int8')\n        customers_df['Active'] = customers_df['Active'].fillna(0).astype('int8')\n        customers_df['club_member_status'] = customers_df['club_member_status'].fillna('UNKNOWN')\n        customers_df['age'] = customers_df['age'].fillna(customers_df['age'].mean()).astype('int8')\n        customers_df['fashion_news_frequency'] = (\n            customers_df['fashion_news_frequency']\n            .replace('None', 'NONE')\n            .replace(np.nan, 'NONE')\n        )\n        return customers_df\n\n    def get(self):\n        output = (\n            self.customers_df[filter(lambda x: x != 'postal_code', customers_df.columns)]\n            .set_index('customer_id')\n        )\n        return output","metadata":{"execution":{"iopub.status.busy":"2022-02-25T18:41:40.970570Z","iopub.execute_input":"2022-02-25T18:41:40.971403Z","iopub.status.idle":"2022-02-25T18:41:40.979187Z","shell.execute_reply.started":"2022-02-25T18:41:40.971365Z","shell.execute_reply":"2022-02-25T18:41:40.978588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ArticlesFeatures(UserFeatures):\n    \"\"\"\n    returns article features: whether category appears in top categories\n    \"\"\"\n    def __init__(self, transactions_df, articles, topk = 10):\n        self.merged_df = transactions_df.merge(articles, on = ('article_id'))\n        self.articles = articles\n        self.topk = topk\n    \n    def get(self):\n        output_df = None\n\n        for col in tqdm(self.articles.columns, desc = 'extracting features'):\n            if 'name' in col:\n                if output_df is None:\n                    output_df = self.aggregate_topk(self.merged_df, col, self.topk)\n                else:\n                    intermediate_out = self.aggregate_topk(self.merged_df, col, self.topk)\n                    output_df = output_df.merge(intermediate_out, on = ('customer_id'))\n        return output_df\n\n    def return_value_counts(self, df, column_name, k):\n        value_counts = df[column_name].value_counts()[:k].index\n        value_counts = list(map(lambda x: x[1], value_counts))\n        return value_counts\n\n    def aggregate_topk(self, merged_df, column_name, k):\n        grouped_df_indx = merged_df.groupby('customer_id')\n        grouped_df = merged_df.groupby('customer_id', as_index = False)\n        \n        topk_values = self.return_value_counts(grouped_df_indx, column_name, k)\n        #how many transactions appears in top category(column)\n        n_top_k = (\n            grouped_df[column_name]\n            .agg({\n                f'top_{column_name}_{i}': lambda x: sum(x == k) for i, k in enumerate(topk_values)\n            })\n            .set_index('customer_id')\n            .astype('int16')\n        )\n        return n_top_k","metadata":{"execution":{"iopub.status.busy":"2022-02-25T18:47:04.804833Z","iopub.execute_input":"2022-02-25T18:47:04.805551Z","iopub.status.idle":"2022-02-25T18:47:04.816839Z","shell.execute_reply.started":"2022-02-25T18:47:04.805507Z","shell.execute_reply":"2022-02-25T18:47:04.816241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UserFeaturesCollector:\n    \"\"\"\n    collect all features and aggregate them\n    \"\"\"\n    @staticmethod\n    def collect(features: Union[List[UserFeatures], List[str]], **kwargs) -> pd.DataFrame:\n        output_df = None\n\n        for feature in tqdm(features):\n            if isinstance(feature, UserFeatures):\n                feature_out = feature.get(**kwargs)\n            if isinstance(feature, str):\n                try:\n                    feature_out = pd.read_csv(feature)\n                except:\n                    feature_out = pd.read_parquet(feature)\n\n            if output_df is None:\n                output_df = feature_out\n            else:\n                output_df = output_df.merge(feature_out, on = ('customer_id'))\n        return output_df","metadata":{"execution":{"iopub.status.busy":"2022-02-25T18:48:18.180814Z","iopub.execute_input":"2022-02-25T18:48:18.181139Z","iopub.status.idle":"2022-02-25T18:48:18.190987Z","shell.execute_reply.started":"2022-02-25T18:48:18.181101Z","shell.execute_reply":"2022-02-25T18:48:18.190319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For simplicity let's take only first 10k transactions.","metadata":{}},{"cell_type":"code","source":"user_features = UserFeaturesCollector.collect([\n    AggrFeatures(transactions_train.iloc[:10_000]),\n    CountFeatures(transactions_train.iloc[:10_000], 3),\n    CustomerFeatures(customers_df),\n    ArticlesFeatures(transactions_train.iloc[:10_000], articles_df, 3),\n])","metadata":{"execution":{"iopub.status.busy":"2022-02-25T18:21:41.583034Z","iopub.execute_input":"2022-02-25T18:21:41.583274Z","iopub.status.idle":"2022-02-25T18:21:45.695867Z","shell.execute_reply.started":"2022-02-25T18:21:41.583247Z","shell.execute_reply":"2022-02-25T18:21:45.694516Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_features.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Save to parquet because it allocates less memory.","metadata":{}},{"cell_type":"code","source":"user_features.to_parquet('user_features.parquet')","metadata":{},"execution_count":null,"outputs":[]}]}