{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# creates \"atomic files\" for recbole","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\ndf_item = pd.read_csv(r\"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\", dtype={'article_id': 'str'})\ndf_phrase_embeddings = pd.read_csv('../input/handmarticledescriptionembeddings/phrase_embeddings.csv').drop(columns=['article_id', 'Unnamed: 0'])\ndf_inter = pd.read_csv(r\"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\", \n                 dtype={'article_id': 'str'})","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-23T07:10:59.035249Z","iopub.execute_input":"2022-04-23T07:10:59.035583Z","iopub.status.idle":"2022-04-23T07:11:07.477350Z","shell.execute_reply.started":"2022-04-23T07:10:59.035497Z","shell.execute_reply":"2022-04-23T07:11:07.475984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_item = df_item.drop(columns = ['product_type_name', 'graphical_appearance_name', 'colour_group_name', 'perceived_colour_value_name',\n                        'perceived_colour_master_name', 'index_name', 'index_group_name', 'section_name', \n                        'garment_group_name', 'prod_name', 'department_name', 'detail_desc'])","metadata":{"execution":{"iopub.status.busy":"2022-04-23T07:11:07.478154Z","iopub.status.idle":"2022-04-23T07:11:07.478932Z","shell.execute_reply.started":"2022-04-23T07:11:07.478595Z","shell.execute_reply":"2022-04-23T07:11:07.478663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## items + phrase encodings","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\nx = df_phrase_embeddings.values\nx_scaled = StandardScaler().fit_transform(x)\nx_pca = PCA(n_components=0.5, svd_solver='full').fit_transform(x_scaled)\ndf_phrase = pd.DataFrame(x_pca)\ndf_phrase.head()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-04-23T07:11:07.483103Z","iopub.status.idle":"2022-04-23T07:11:07.483416Z","shell.execute_reply.started":"2022-04-23T07:11:07.483250Z","shell.execute_reply":"2022-04-23T07:11:07.483267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_phrase.to_csv('phrase_embeddings_pca.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_item = pd.concat([df_item, df_phrase], axis=1)\ndf_item.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-04-23T07:11:07.484294Z","iopub.status.idle":"2022-04-23T07:11:07.484618Z","shell.execute_reply.started":"2022-04-23T07:11:07.484435Z","shell.execute_reply":"2022-04-23T07:11:07.484451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = df_item.rename(\n    columns={'article_id': 'item_id:token', 'product_code': 'product_code:token', 'product_type_no': 'product_type_no:float',\n             'product_group_name': 'product_group_name:token_seq', 'graphical_appearance_no': 'graphical_appearance_no:token', \n             'colour_group_code': 'colour_group_code:token', 'perceived_colour_value_id': 'perceived_colour_value_id:token', \n             'perceived_colour_master_id': 'perceived_colour_master_id:token', 'department_no': 'department_no:token', \n             'index_code': 'index_code:token', 'index_group_no': 'index_group_no:token', 'section_no': 'section_no:token', \n             'garment_group_no': 'garment_group_no:token',\n             **{i: f'{i}:float' for i in range(df_phrase.shape[1])}})\ntemp.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-23T07:11:07.485675Z","iopub.status.idle":"2022-04-23T07:11:07.486031Z","shell.execute_reply.started":"2022-04-23T07:11:07.485848Z","shell.execute_reply":"2022-04-23T07:11:07.485867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/hm_data\ntemp.to_csv(r'/kaggle/working/hm_data/hm_data.item', index=False, sep='\\t')","metadata":{"execution":{"iopub.status.busy":"2022-04-23T07:11:07.487009Z","iopub.status.idle":"2022-04-23T07:11:07.487293Z","shell.execute_reply.started":"2022-04-23T07:11:07.487140Z","shell.execute_reply":"2022-04-23T07:11:07.487156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## interactions","metadata":{}},{"cell_type":"code","source":"df_inter['t_dat'] = pd.to_datetime(df_inter['t_dat'], format=\"%Y-%m-%d\")\ndf_inter['timestamp'] = df_inter.t_dat.values.astype(np.int64) // 10 ** 9\ndf_inter.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = df_inter[df_inter['timestamp'] > 1585620000][['customer_id', 'article_id', 'timestamp']].rename(\n    columns={'customer_id': 'user_id:token', 'article_id': 'item_id:token', 'timestamp': 'timestamp:float'})\ntemp","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp.to_csv('/kaggle/working/hm_data/hm_data.inter', index=False, sep='\\t')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# default rec","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0 = pd.read_csv('../input/hm-pre-recommendation/submissio_byfone_chris.csv').sort_values('customer_id').reset_index(drop=True)\nsub1 = pd.read_csv('../input/hm-pre-recommendation/submission_trending.csv').sort_values('customer_id').reset_index(drop=True)\nsub2 = pd.read_csv('../input/hm-pre-recommendation/submission_exponential_decay.csv').sort_values('customer_id').reset_index(drop=True)\n\nsub0.shape, sub1.shape, sub2.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0.columns = ['customer_id', 'prediction0']\nsub0['prediction1'] = sub1['prediction']\nsub0['prediction2'] = sub2['prediction']\ndel sub1, sub2","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cust_blend(dt, W = [1,1,1]):\n    #Global ensemble weights\n    #W = [1.15,0.95,0.85]\n    \n    #Create a list of all model predictions\n    REC = []\n    REC.append(dt['prediction0'].split())\n    REC.append(dt['prediction1'].split())\n    REC.append(dt['prediction2'].split())\n    \n    #Create a dictionary of items recommended. \n    #Assign a weight according the order of appearance and multiply by global weights\n    res = {}\n    for M in range(len(REC)):\n        for n, v in enumerate(REC[M]):\n            if v in res:\n                res[v] += (W[M]/(n+1))\n            else:\n                res[v] = (W[M]/(n+1))\n    \n    # Sort dictionary by item weights\n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n    \n    # Return the top 12 itens only\n    return ' '.join(res[:12])\n\nsub0['prediction'] = sub0.apply(cust_blend, W = [1.05,1.00,0.95], axis=1)\nsub0.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sub0['prediction0']\ndel sub0['prediction1']\ndel sub0['prediction2']\nsub0.to_csv(f'submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}