{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport plotly.graph_objects as go\nfrom plotly.offline import iplot, init_notebook_mode\ninit_notebook_mode(connected=True)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.offline import iplot, init_notebook_mode\ninit_notebook_mode(connected=True)\nimport plotly_express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode\nimport plotly.io as pio\nfrom plotly.subplots import make_subplots\n# setting default template to plotly_white for all visualizations\npio.templates.default = \"plotly_white\"\n%matplotlib inline\nimport gc\n\nfrom colorama import Fore, Back, Style\n\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\nc_ = Fore.CYAN\nres = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings('ignore')\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\nimport folium\nimport matplotlib.dates as mdates\nfrom datetime import datetime\nfrom mpl_toolkits.axes_grid1 import make_axes_locatable\nfrom matplotlib.offsetbox import AnchoredText\nimport matplotlib\nYELLOVE = '#fdb913'\n\nimport os\nDATE_TO_FILTER = '2020-09-01'","metadata":{"_uuid":"a40fea84-b2ef-4848-808f-a07d4edbd37f","_cell_guid":"96382db3-5f7a-4cf3-bcef-28416eb14d2d","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv', index_col=None)\narticles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv', dtype={'article_id': str},index_col=None)\ntrans = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', dtype={'article_id': str}, parse_dates=['t_dat'])\nsubmission = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv', index_col=None)\n#/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\ncust.shape, articles.shape, trans.shape, submission.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trans.head(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_features = ['article_id','prod_name','product_type_name','product_group_name','perceived_colour_master_name','department_name','section_name','detail_desc']\narticles[article_features]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = trans[trans['t_dat'] >= DATE_TO_FILTER].reset_index(drop=True)\ndf.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =  df.groupby(['customer_id','article_id']).agg({'price': ['mean'],'sales_channel_id' : ['count']}).reset_index()\ndf.columns = df.columns.get_level_values(0)\ndf.rename(columns={'sales_channel_id':'count'},inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['count'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zipcode_map = dict(zip(cust.customer_id, cust.postal_code))\ndf['customer_zipcode'] = df['customer_id'].apply(lambda x: zipcode_map[x])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop_duplicates(\n  subset = ['customer_id', 'article_id','price'],\n  keep = 'last')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(\n    articles,\n    on='article_id',\n    how='left'\n)[['customer_id','article_id','price','count','customer_zipcode','prod_name','product_type_name','product_group_name','perceived_colour_master_name','department_name','section_name','detail_desc']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(\n    cust,\n    on='customer_id',\n    how='left'\n)[['customer_id','age','article_id', 'price', 'count', 'customer_zipcode',\n       'prod_name', 'product_type_name', 'product_group_name',\n       'perceived_colour_master_name', 'department_name', 'section_name',\n       'detail_desc']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['age'].plot(kind='hist')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['age'].describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Pipeline","metadata":{}},{"cell_type":"code","source":"from sklearn.base import BaseEstimator\n\n# 1. First Transformation - Filling missing Age with median age from customers zip code\nclass FillMissingCustomerAge(BaseEstimator):\n    def __init__(self):\n        pass\n    \n    def fit(self, X, y=None):\n        return self\n    \n    def transform(self, X):\n        self._zip_cd_med_age_df = X.groupby(['customer_zipcode'])['age'].median().reset_index()\n        self._zip_cd_med_age_df['age'] = self._zip_cd_med_age_df['age'].fillna(self._zip_cd_med_age_df['age'].median())\n        self._zip_code_medag_map = dict(zip(self._zip_cd_med_age_df['customer_zipcode'], self._zip_cd_med_age_df['age']))\n        X['age'] = X.apply(lambda row: self._zip_code_medag_map[row['customer_zipcode']] if np.isnan(row['age']) else row['age'],axis=1)\n        #X['age'] = X.apply(lambda row: zma_map[row['customer_zipcode']] if np.isnan(row['age']) else row['age'],axis=1)\n        return X\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#2. Convert Age into categorical atrribute with quantile cuts.\nclass CustomerAgeGroupAdder(BaseEstimator):\n    def __init__(self, q=5, bin_labels=['Group1','Group2','Group3','Group4','Group5'], print_threshold = True):\n        self._q = q\n        self._bin_labels = bin_labels\n        self._print_threshold = print_threshold\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def transform(self, X):\n        X['customer_age_group'] = pd.qcut(X['age'],q=self._q,labels=self._bin_labels)\n        if self._print_threshold == True:\n            results, bin_edges = pd.qcut(X['age'],q=self._q,labels=self._bin_labels,retbins=True)\n            results_table = pd.DataFrame(zip(bin_edges, self._bin_labels),\n                            columns=['Threshold', 'Tier'])\n            #print(\"Age threshold to groups : \",results_table.to_dict('list'))\n            print(\"Age threshold to groups :\\n{}\\n{}\".format(bin_edges, self._bin_labels))\n            \n        return X","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#3. If product description is empty then copy product name\nclass FillProductDesc(BaseEstimator):\n    def __init__(self):\n        pass\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def transform(self, X):\n        X['detail_desc'] = X['detail_desc'].astype(str)\n        X['detail_desc'] = X.apply(lambda row: row['prod_name'] if row['detail_desc'] == 'nan' else row['detail_desc'], axis=1)\n        return X","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom scipy.sparse import csr_matrix, hstack, save_npz\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\n#4. Encode data and vectorize text attributes. Final step of the data pre-processing\nclass EncodeData(BaseEstimator):\n    def __init__(self, inference_data=False, ohe_cols = [\"article_id\", \"customer_id\", \"customer_age_group\", \"product_type_name\", \"perceived_colour_master_name\", \"department_name\", \"section_name\"],\n                word2vec_cols = ['prod_name','detail_desc'],\n                min_df = 2):\n        self._ohe_cols = ohe_cols\n        self._word2vec_cols = word2vec_cols\n        self._inference_data = inference_data\n        self._min_df = min_df\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def transform(self, X, inference_data=False):\n        enc = OneHotEncoder(handle_unknown=\"ignore\")\n        onehot_cols = [\"article_id\", \"customer_id\", \"customer_age_group\", \"product_type_name\", \"perceived_colour_master_name\", \"department_name\", \"section_name\"]\n        ohe_output = enc.fit_transform(X[onehot_cols])\n        y = None\n        \n        w2v_lst = []\n        for w2v_col in self._word2vec_cols:\n            tfid_vec = TfidfVectorizer(min_df=self._min_df)\n            unique_txt = X[w2v_col].unique()\n            tfid_vec.fit(unique_txt)\n            w2v_lst.append(tfid_vec.transform(X[w2v_col]))\n        \n        lst = []\n        lst.append(ohe_output)\n        for item in w2v_lst:\n            lst.append(item)\n\n        row = range(len(X))\n        col = [0] * len(X)\n        unit_price = csr_matrix((X[\"price\"].values, (row, col)), dtype=\"float32\")\n\n        lst.append(unit_price)\n            \n        x = hstack(lst, format=\"csr\", dtype=\"float32\")\n\n        if self._inference_data == False: #Extract target variable for only training data\n            y = X[\"count\"].values.astype(\"float32\")\n        return x, y\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df['price'].plot(kind='hist')\n#from sklearn.preprocessing import StandardScaler\n#StandardScaler().fit_transform([df['price']])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\n\nrecomm_pipeline = Pipeline([\n    ('FillingMissingAge',FillMissingCustomerAge()),\n    ('CutAgeToCategorical',CustomerAgeGroupAdder()),\n    ('FillProductDesc',FillProductDesc()),\n    ('EncodeVectorizeData',EncodeData(inference_data = False, min_df = 0.1))\n]\n)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y = recomm_pipeline.fit_transform(df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nX_train.shape, X_test.shape, y_train.shape, y_test.shape\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"hm_train.csv\", index=False)\nsave_npz(\"X_hm_train.npz\", X_train)\nsave_npz(\"X_hm_test.npz\", X_test)\nnp.savez(\"y_hm_train.npz\", y_train)\nnp.savez(\"y_hm_test.npz\", y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare prediction data","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(['article_id'])['count'].sum().reset_index().sort_values(by=\"count\",ascending=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['sale_price'] = df['count']*df['price']\nNUM_TOP_PRODUCTS = 5\ntop_prods = list(df.groupby(['article_id'])['sale_price'].sum().reset_index().sort_values(by=\"sale_price\",ascending=False).head(NUM_TOP_PRODUCTS)['article_id'])\ntop_prods","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df.customer_id.unique())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df = pd.DataFrame({'customer_id':df.customer_id.unique()})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cus = ['A','B','C','D']\nprd = ['123','345','678','901']\n\ncustomer = []\nproduct = []\nfor p in prd:\n    for c in cus:\n        customer.append(c)\n        product.append(p)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame({'cust':customer, 'prod':product})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customer = []\nproduct = []\nfor p in top_prods:\n    for c in df.customer_id.unique():\n        customer.append(c)\n        product.append(p)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df = pd.DataFrame({'customer_id':customer, 'article_id':product})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df['customer_zipcode'] = inference_df['customer_id'].apply(lambda x: zipcode_map[x])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_price = df.groupby('article_id')['price'].mean().reset_index()\narticle_price_map = dict(zip(article_price['article_id'], article_price['price']))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df['price'] = inference_df.apply(lambda row: article_price_map[row['article_id']], axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df = inference_df.merge(\n    articles,\n    on='article_id',\n    how='left'\n)[['customer_id','article_id','price','customer_zipcode','prod_name','product_type_name','product_group_name','perceived_colour_master_name','department_name','section_name','detail_desc']]\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df = inference_df.merge(\n    cust,\n    on='customer_id',\n    how='left'\n)[['customer_id','age','article_id', 'price', 'customer_zipcode',\n       'prod_name', 'product_type_name', 'product_group_name',\n       'perceived_colour_master_name', 'department_name', 'section_name',\n       'detail_desc']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recomm_pipeline = Pipeline([\n    ('FillingMissingAge',FillMissingCustomerAge()),\n    ('CutAgeToCategorical',CustomerAgeGroupAdder()),\n    ('FillProductDesc',FillProductDesc()),\n    ('EncodeVectorizeData',EncodeData(inference_data = True, min_df = 0.1))\n]\n)\nX_inference , _ = recomm_pipeline.fit_transform(inference_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_npz(\"X_hm_inference.npz\", X_inference)\ninference_df.to_csv(\"hm_inference.csv\", index=False)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[['customer_id', 'age', 'article_id', 'price', 'count',\n       'customer_zipcode', 'prod_name', 'product_type_name',\n       'product_group_name', 'perceived_colour_master_name', 'department_name',\n       'section_name', 'detail_desc', 'customer_age_group']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recomm_pipeline = Pipeline([\n    ('FillingMissingAge',FillMissingCustomerAge()),\n    ('CutAgeToCategorical',CustomerAgeGroupAdder()),\n    ('FillProductDesc',FillProductDesc()),\n    ('EncodeVectorizeData',EncodeData(inference_data = False, min_df = 0.1))\n]\n)\nX_tdata, y = recomm_pipeline.fit_transform(df)\nX_tdata","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.loc[articles['article_id'].isin(top_prods)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}