{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%matplotlib inline\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom IPython import display\nplt.rcParams.update({'figure.figsize': [8,8]})\nplt.rcParams['axes.facecolor'] = 'white'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-14T14:45:28.951373Z","iopub.execute_input":"2022-09-14T14:45:28.951994Z","iopub.status.idle":"2022-09-14T14:45:28.988992Z","shell.execute_reply.started":"2022-09-14T14:45:28.951875Z","shell.execute_reply":"2022-09-14T14:45:28.987927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# H&M 個人化時尚產品推薦\n\n時尚產品的推薦:\n* 性別差異以及中性服飾\n* 嬰兒服飾推薦\n* 季節性商品\n* 會重複購買語不會重複購買的商品型態\n","metadata":{}},{"cell_type":"code","source":"import glob\nimport os\nimport cv2\nimport gc\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nos.environ['TRIDENT_BACKEND'] = 'pytorch'\n!pip uninstall tridentx -y\n!pip install ../input/trident/tridentx-0.7.5-py3-none-any.whl --upgrade\nimport trident as T\nfrom trident import *","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:45:28.993939Z","iopub.execute_input":"2022-09-14T14:45:28.994823Z","iopub.status.idle":"2022-09-14T14:45:50.144111Z","shell.execute_reply.started":"2022-09-14T14:45:28.994782Z","shell.execute_reply":"2022-09-14T14:45:50.142811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_articles=pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_articles['text']=df_articles['index_name']+', '+df_articles['section_name']+', '+df_articles['detail_desc']+', '+df_articles['colour_group_name']\ndf_articles.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:37:34.407903Z","iopub.execute_input":"2022-09-14T14:37:34.408623Z","iopub.status.idle":"2022-09-14T14:37:36.107386Z","shell.execute_reply.started":"2022-09-14T14:37:34.408581Z","shell.execute_reply":"2022-09-14T14:37:36.106442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_ids=df_articles['article_id'].unique()\narticle_id_mapping=OrderedDict()\narticle_id_mapping[0]=0\nfor i, id in enumerate(article_ids):\n    article_id_mapping[id]=i+1\n\nprint(len(article_id_mapping))\nprint(list(article_id_mapping.items())[:10])\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:37:36.112963Z","iopub.execute_input":"2022-09-14T14:37:36.115167Z","iopub.status.idle":"2022-09-14T14:37:36.250594Z","shell.execute_reply.started":"2022-09-14T14:37:36.115129Z","shell.execute_reply":"2022-09-14T14:37:36.249340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 處理訓練數據\n\n讀取df_train，更新時間欄位以及加入時間衍生變數  \n基於df_train產生customer_ids清單以及建立索引的對照表\n","metadata":{}},{"cell_type":"code","source":"df_train=pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ndf_train['t_dat'] = pd.to_datetime(df_train['t_dat'])\n#加入季節性以及星期己的概念\ndf_train['t_days']=df_train['t_dat'].dt.day_of_year\ndf_train['t_weekday']=df_train['t_dat'].dt.weekday.astype('category')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:37:36.255203Z","iopub.execute_input":"2022-09-14T14:37:36.257566Z","iopub.status.idle":"2022-09-14T14:38:49.538481Z","shell.execute_reply.started":"2022-09-14T14:37:36.257525Z","shell.execute_reply":"2022-09-14T14:38:49.537333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"首先基於完整的交易紀錄產生customer_id對索引的轉換字典。","metadata":{}},{"cell_type":"code","source":"#產生customer_id轉索引的字典\ncustomer_ids=sorted(df_train['customer_id'].unique())\ncustomer_id_mapping = {id:i for i, id in enumerate(customer_ids)}","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:38:49.539857Z","iopub.execute_input":"2022-09-14T14:38:49.540473Z","iopub.status.idle":"2022-09-14T14:38:57.906826Z","shell.execute_reply.started":"2022-09-14T14:38:49.540435Z","shell.execute_reply":"2022-09-14T14:38:57.905717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"為了可以建模當下驗證模型的有效性，由於數據量頗大，kaggle機器會無法承擔，一開始僅使用到2019/8/31~2020/8/31前(不含)的數據來建構用戶embedded(訓練歷史資料，*df_train_history*)，用後3週成交紀錄(*transactions_last*)做為正樣本來確保高召回(將三周內的銷售都納入計算，以獲取高召回)，後一週的數據(*transactions_l1w*)用來進行精排。","metadata":{}},{"cell_type":"code","source":"#建模用歷史數據\nmask_history= (df_train.t_dat >= pd.to_datetime('2019-08-31')) & (df_train.t_dat < pd.to_datetime('2020-08-31'))\ndf_train_history = df_train[mask_history]\n\n#前1個月\nmask_p1m = (df_train.t_dat >= pd.to_datetime('2020-07-31')) & (df_train.t_dat < pd.to_datetime('2020-08-31'))\ntransactions_p1m = df_train[mask_p1m]\n\n\n#後1週\nmask_l1w = (df_train.t_dat >= pd.to_datetime('2020-08-31')) & (df_train.t_dat < pd.to_datetime('2020-09-07'))\ntransactions_l1w = df_train[mask_l1w]\n#transactions_l2w = df_train[mask_l2w]\n\n#後3週\ntransactions_last = df_train[df_train['t_dat'] >= pd.to_datetime('2020-08-31')]\n\npickle_it('./transactions_p1m.pkl',transactions_p1m)\npickle_it('./transactions_l1w.pkl',transactions_l1w)\npickle_it('./transactions_last.pkl',transactions_last)\n\n#刪除以節省記憶體\ndel df_train\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:38:57.908616Z","iopub.execute_input":"2022-09-14T14:38:57.909017Z","iopub.status.idle":"2022-09-14T14:39:00.836785Z","shell.execute_reply.started":"2022-09-14T14:38:57.908978Z","shell.execute_reply":"2022-09-14T14:39:00.835265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#index_group_name去重複的值列表\nindex_group_names=df_articles['index_group_name'].unique()\nindex_group_names_mapping=OrderedDict()\nindex_group_names_mapping['0']=0\nfor i, id in enumerate(index_group_names):\n    index_group_names_mapping[id]=i+1\n\nprint('index_group_name',df_articles['index_group_name'].unique())\nprint('section_name',df_articles['section_name'].unique())\n\n\ndef map_article(article_id):\n    return article_id_mapping[article_id]\n\n#將article_id更換成索引\ndf_articles['article_id']=df_articles['article_id'].apply(map_article)\n\n#輸入article_idx輸出為index_group_names字串\nindex_group_dict=OrderedDict(zip(df_articles['article_id'].to_numpy(),df_articles['index_group_name'].to_numpy()))\nindex_group_dict[0]='0'\n\n#輸入article_idx輸出為index_group_names_idx\nindex_group_index_dict={}\nfor k in index_group_dict.keys():\n    if int(k)==0:\n        index_group_index_dict[0]=0\n    else:\n        index_group_index_dict[k]=index_group_names_mapping[index_group_dict[k]]\n\n\n#key:index_group_name  value:該類別之商品清單\ndf_articles=df_articles.groupby(['index_group_name']).agg({'article_id':lambda x: list(x)})\ndf_articles= df_articles.reset_index()\nindex_group_pools=OrderedDict(zip(df_articles['index_group_name'].to_numpy(),df_articles['article_id'].to_numpy()))\n#print(index_group_dict)\n\n\n# index_group_pools=df_articles.groupby(['index_group_name']).agg({'article_id':lambda x: list(x)})\n\ndel df_articles\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:39:00.838363Z","iopub.execute_input":"2022-09-14T14:39:00.838802Z","iopub.status.idle":"2022-09-14T14:39:01.258677Z","shell.execute_reply.started":"2022-09-14T14:39:00.838761Z","shell.execute_reply":"2022-09-14T14:39:01.257572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"tmp_shopping_history裡面是紀錄了{用戶index: { 購物日期: \\[產品清單\\]}}  \n\nshopping_history裡面是紀錄了{用戶index: \\[購物日星期幾,產品清單\\]}   \n\nuser_preference裡面是紀錄了{用戶index: { index_group_names: 數量}}       \n","metadata":{}},{"cell_type":"markdown","source":"由於pandas的dataframe 遞迴的速度太慢，因此採取直接在dataframe進行apply直接產生需要的欄位，最後再透過series的方式取出個別欄位，組裝成dict這樣會比較快速。首先我們想要整理出每個客戶過去的購物歷程記錄，其中以*customer_id + t_dat*視為單次的購物歷程。","metadata":{}},{"cell_type":"code","source":"sub_history=df_train_history[['customer_id','t_dat','article_id']]\n\ndel df_train_history\ngc.collect()\nsub_history","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:39:01.261814Z","iopub.execute_input":"2022-09-14T14:39:01.262763Z","iopub.status.idle":"2022-09-14T14:39:01.922239Z","shell.execute_reply.started":"2022-09-14T14:39:01.262726Z","shell.execute_reply":"2022-09-14T14:39:01.921289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"由於後面建模都是使用索引值來替代customer_id以及article_id，因此先直接替換掉。同時後續為了建構負樣本，需要知道哪些index_group_name是客戶買過的，因此也同時需要先將購買的article_id對應的index_group_name留存。這裡寫了3個函數透過apply來進行轉換。","metadata":{}},{"cell_type":"code","source":"#直接原地把article_id與customer_id都換成索引\ndef map_article(article_id):\n    return article_id_mapping[article_id]\n\ndef map_customer(cust_id):\n    return customer_id_mapping[cust_id]\n\ndef map_index_group_name(article_id):\n    return index_group_dict[article_id]\n\n\nsub_history['article_id']=sub_history['article_id'].apply(map_article)\nsub_history['customer_id']=sub_history['customer_id'].apply(map_customer)\n#比對買過的類別\nsub_history['index_group_name']=sub_history['article_id'].apply(map_index_group_name)\nsub_history","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:39:01.926212Z","iopub.execute_input":"2022-09-14T14:39:01.926821Z","iopub.status.idle":"2022-09-14T14:39:27.913713Z","shell.execute_reply.started":"2022-09-14T14:39:01.926783Z","shell.execute_reply":"2022-09-14T14:39:27.912556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_bought_items=sub_history.groupby(['customer_id']).agg({'article_id':lambda x: list(set(x))})\nuser_bought_items= user_bought_items.reset_index()\n\nuser_bought_items_dict=OrderedDict(zip(user_bought_items['customer_id'].to_list(),user_bought_items['article_id'].to_list()))\nuser_bought_items.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:39:27.915364Z","iopub.execute_input":"2022-09-14T14:39:27.915780Z","iopub.status.idle":"2022-09-14T14:39:39.281406Z","shell.execute_reply.started":"2022-09-14T14:39:27.915740Z","shell.execute_reply":"2022-09-14T14:39:39.280340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"接下來我們利用 customer_id , t_dat 作為key進行groupby，並且將article_id進行彙總，其彙總方式是將它組成為清單。但是我們還需要針對這個清單做一些改造。  \n首先我們希望傳入這次購物日期星期幾資訊，但又不想要額外弄一個獨立的資料集，所以最簡單的方式是把它塞在article清單中的第一位，因為都是整數值所以問題不大。\n","metadata":{}},{"cell_type":"code","source":"max_items=32+1\ntmp_shopping_history=sub_history.groupby(['customer_id','t_dat']).agg({'article_id':lambda x: list(x)})\ntmp_shopping_history= tmp_shopping_history.reset_index()\n\n\n\ndef make_array(article_id,t_dat):\n    #把序列內產品去重複排序\n    article_id=list(set(article_id))\n    #把weekday塞在第一位\n    article_id.insert(0,t_dat.weekday())\n    if len(article_id)<max_items:\n        #序列透過用0補滿\n        article_id.extend([0]*(max_items-len(article_id)))\n    elif len(article_id)>max_items:\n        #若是過長則截斷\n        article_id=article_id[:max_items]\n    return article_id\n\ntmp_shopping_history['array'] = tmp_shopping_history.apply(lambda x: make_array(x.article_id, x.t_dat), axis=1)\ntmp_shopping_history=tmp_shopping_history.drop(['t_dat', 'article_id'], axis=1)\n\ntmp_shopping_history","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:39:39.282845Z","iopub.execute_input":"2022-09-14T14:39:39.283301Z","iopub.status.idle":"2022-09-14T14:42:13.946801Z","shell.execute_reply.started":"2022-09-14T14:39:39.283263Z","shell.execute_reply":"2022-09-14T14:42:13.945848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"前一步驟的匯總已經構成了每次購物歷程，這一次再作彙總則是建構一個客戶的歷史購物歷程集合。","metadata":{}},{"cell_type":"code","source":"\ntmp_shopping_history=tmp_shopping_history.groupby(['customer_id']).agg({'array':lambda x: list(x)})\n\ntmp_shopping_history= tmp_shopping_history.reset_index()\n\ndef make_array(array):\n    array=np.asarray(array).astype(np.int64)\n    if ndim(array)<2:\n        array=np.expand_dims(array,0)\n    return array\n\ntmp_shopping_history['array']=tmp_shopping_history['array'].apply(make_array)\ntmp_shopping_history\n\n\ntraining_sample={}\n\ntraining_sample['tmp_shopping_history']=tmp_shopping_history\n\npickle_it('./training_sample.pkl',training_sample)\nprint('pickle finished !')\n\nshopping_history=OrderedDict(zip(tmp_shopping_history['customer_id'].to_list(),tmp_shopping_history['array'].to_list()))\ndel tmp_shopping_history\n\ngc.collect()\n\n\ntraining_sample['shopping_history']=shopping_history\n\npickle_it('./training_sample.pkl',training_sample)\nprint('pickle finished !')\n\ndel training_sample['tmp_shopping_history']","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:42:13.948501Z","iopub.execute_input":"2022-09-14T14:42:13.948879Z","iopub.status.idle":"2022-09-14T14:43:28.404579Z","shell.execute_reply.started":"2022-09-14T14:42:13.948842Z","shell.execute_reply":"2022-09-14T14:43:28.403451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"在這邊我們把有買過的index_group_name彙總成清單，然後根據完整的index_group_name列表扣掉已經買過的，這樣就可以得到用戶不想買的類別，這主要是用在產生負樣本時使用。","metadata":{}},{"cell_type":"code","source":"user_preference=sub_history.groupby(['customer_id']).agg({'index_group_name':lambda x: list(set(x))})\nuser_preference= user_preference.reset_index()\nuser_preference_dict=OrderedDict(zip(user_preference['customer_id'],user_preference['index_group_name']))\ndel sub_history\ngc.collect()\n\n#根據完整index_group_names扣掉有買過的\ndef remove_preference(index_group_name):\n    global index_group_names\n    return [item  for item in index_group_names if item not in index_group_name and item!='Divided']\n\n\nuser_preference['index_group_name']=user_preference['index_group_name'].apply(remove_preference)\nuser_preference= user_preference.reset_index()\nuser_preference.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:43:28.406257Z","iopub.execute_input":"2022-09-14T14:43:28.406927Z","iopub.status.idle":"2022-09-14T14:43:41.013103Z","shell.execute_reply.started":"2022-09-14T14:43:28.406887Z","shell.execute_reply":"2022-09-14T14:43:41.011953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#查詢該用戶不想買的分類\nuser_not_preference_dict=OrderedDict(zip(user_preference['customer_id'],user_preference['index_group_name']))\ndel user_preference\ngc.collect()\n\n\ntraining_sample['user_preference_dict']=user_preference_dict\ntraining_sample['user_not_preference_dict']=user_not_preference_dict\ntraining_sample['shopping_history']=shopping_history\npickle_it('./training_sample.pkl',training_sample)\nprint('pickle finished !')","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:43:41.014854Z","iopub.execute_input":"2022-09-14T14:43:41.015270Z","iopub.status.idle":"2022-09-14T14:43:53.024853Z","shell.execute_reply.started":"2022-09-14T14:43:41.015232Z","shell.execute_reply":"2022-09-14T14:43:53.023665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customers=pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')\n#年齡遺漏值填-1\ndf_customers=df_customers.fillna({'age': -1})\n\nage_dict=OrderedDict(zip(df_customers['customer_id'],df_customers['age']))\ndel df_customers\nuser_age_dict={}\n#因為之前customer_id_mapping是基於df_train做出來的，因此與age_dict全體基數不一樣\nfor k,v in customer_id_mapping.items():\n    if k in age_dict:\n        user_age_dict[v]=age_dict[k]\n    else:\n        print(k)\n\ntraining_sample['user_age_dict']=user_age_dict\npickle_it('./training_sample.pkl',training_sample)\nprint('pickle finished !')","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:43:53.026528Z","iopub.execute_input":"2022-09-14T14:43:53.026895Z","iopub.status.idle":"2022-09-14T14:44:10.531783Z","shell.execute_reply.started":"2022-09-14T14:43:53.026858Z","shell.execute_reply":"2022-09-14T14:44:10.530725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport gc\n\n\n# torch.cuda.synchronize()\n# torch.cuda.empty_cache()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:44:10.533319Z","iopub.execute_input":"2022-09-14T14:44:10.533988Z","iopub.status.idle":"2022-09-14T14:44:12.891587Z","shell.execute_reply.started":"2022-09-14T14:44:10.533947Z","shell.execute_reply":"2022-09-14T14:44:12.890267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 建構召回模型與粗排的訓練樣本\n召回模型的意義在於縮減樣本，但是要盡量保留所有可能的商品，因此我們使用較長的時窗，將這時間窗所有商品都視為可能商品。  \n\n建模樣本為(使用者,候選商品,是否購買)三元組，其中使用者索引輸入模型後會置換為這個人過去的購物歷程綜合的Embedding，候選商品Embedding則是基於圖像、分類以及文字特徵融合。\n\n","metadata":{}},{"cell_type":"code","source":"#取出最近一個月交易資料\ntransactions_p1m=unpickle('./transactions_p1m.pkl')\n\n\npopular_items=transactions_p1m.groupby('article_id').agg({'customer_id':['nunique']})\npopular_items.reset_index(inplace=True)\npopular_items.columns=['article_id','cnt']\nprint(len(popular_items))\npopular_items.head()\n\nmask=popular_items['cnt']<10\n\nless_popular_items=popular_items[mask]\nprint(len(less_popular_items))\nless_popular_items.head()\n\nless_popular_items_ids=less_popular_items['article_id'].unique()\nless_popular_items_indexs=[article_id_mapping[article_id] for article_id in less_popular_items_ids]\nprint(len(less_popular_items_indexs))\nprint(len(article_id_mapping))\nprint(less_popular_items_indexs[:100])\n\ndel transactions_p1m","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:44:12.893344Z","iopub.execute_input":"2022-09-14T14:44:12.894701Z","iopub.status.idle":"2022-09-14T14:44:13.877600Z","shell.execute_reply.started":"2022-09-14T14:44:12.894662Z","shell.execute_reply":"2022-09-14T14:44:13.876528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hot_popular_items=popular_items[popular_items['cnt']>150]\nhot_popular_items_ids=hot_popular_items['article_id'].unique()\nhot_popular_items_indexs=[article_id_mapping[article_id] for article_id in hot_popular_items_ids]\n\nhot_index_group_pools=OrderedDict()\nfor k,v in index_group_pools.items():\n    if 'Baby' in k:\n        hot_index_group_pools[k]=v\n    else:\n        hot_index_group_pools[k]=[idx for idx in v if idx in hot_popular_items_indexs]\n    print(k,len(hot_index_group_pools[k]))\n\ndel popular_items","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:44:13.879049Z","iopub.execute_input":"2022-09-14T14:44:13.879670Z","iopub.status.idle":"2022-09-14T14:44:15.563584Z","shell.execute_reply.started":"2022-09-14T14:44:13.879631Z","shell.execute_reply":"2022-09-14T14:44:15.562452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n### 負樣本\n在準備建模樣本時，由於客戶的購買紀錄是稀疏的(只購買了集少的商品，大部分商品沒有購買)，因此是天生不均衡的樣本分布，當我們根據未來3個月的交易狀況的三元組建構了正樣本之外，也會需要準備對應的負樣本，為了避免不均衡，負樣本以抽樣的方式，不會遵循自然的分布，而控制至在正樣本的3.5~10的倍數。至於負樣本的挑選，我採取以下原則:\n\n根據用戶沒有買過的index_group_name內進行抽樣  \n\n根據商品流行程度，從未購買且低流行度的商品內抽樣","metadata":{}},{"cell_type":"code","source":"import random\nusers=[]\ncandidates=[]\nbuys=[]\n\ntest_users=[]\ntest_candidates=[]\ntest_buys=[]\n#產生負樣本\n\ndef get_negative_sample(user_index):\n    not_prefer=[]\n    #基於交互程度的負採樣(根據該用戶從來沒有買過的類別中選取商品做為負樣本，選擇較熱賣的，對於模型影響力較大)\n    if random.random()<0.7:\n        #之前沒出現過的客戶\n        if user_index not in user_not_preference_dict:\n            return None\n        else:\n            not_prefer=user_not_preference_dict[user_index]\n            if len(not_prefer)==0:\n                return None\n\n            return random.choice(hot_index_group_pools[random.choice(not_prefer)])\n    else:\n        #基於流行度的負採樣\n        if user_index in user_bought_items_dict:\n            bought_items=user_bought_items_dict[user_index]\n            item=random.choice(less_popular_items_indexs)\n            if item not in bought_items:\n                return item\n            else:\n                return None\n        else:\n            return random.choice(less_popular_items_indexs)\n            \n            \n        \ntransactions_last=unpickle('./transactions_last.pkl')\n#從最後3週擴大範圍作為召回的主要範圍\nfor i,x in enumerate(tqdm(zip(transactions_last['customer_id'], transactions_last['article_id']),total=len(transactions_last))):\n    cust_id, art_id = x\n    \n    #必須要是訓練數據集曾經出現過的人\n    if customer_id_mapping[cust_id] in shopping_history:\n        if random.random()>0.8:\n                    #成交，正樣本\n            test_users.append(customer_id_mapping[cust_id])\n            test_candidates.append(article_id_mapping[art_id])\n            test_buys.append(1)\n            #採集4倍負樣本\n            for k in range(10):\n                negative=get_negative_sample(customer_id_mapping[cust_id])\n                if negative is not None:\n                    test_users.append(customer_id_mapping[cust_id])\n                    test_candidates.append(negative)\n                    test_buys.append(0)\n        else:\n            #成交，正樣本\n            users.append(customer_id_mapping[cust_id])\n            candidates.append(article_id_mapping[art_id])\n            buys.append(1)\n            #採集4倍負樣本\n            for k in range(4):\n                negative=get_negative_sample(customer_id_mapping[cust_id])\n                if negative is not None:\n                    users.append(customer_id_mapping[cust_id])\n                    candidates.append(negative)\n                    buys.append(0)\n                    \n                    \n                    \nexclude_list=transactions_last['customer_id'].to_list()\nexclude_dict={}\nfor cust_id in exclude_list:\n    exclude_dict[customer_id_mapping[cust_id]]=None\n\nno_show_shoppers=list(shopping_history.keys())\nprint('exclude_list',len(exclude_list),'no_show_shoppers',len(no_show_shoppers))\nno_show_shoppers=[k for k in tqdm(no_show_shoppers,total=len(no_show_shoppers)) if k not in exclude_dict]\n\n\n#比對出建模期間有購買紀錄，但是未來三週卻沒出現的，其購買行為未知，所以斟酌抽樣，且不會有正樣本\nfor user_idx in  tqdm(no_show_shoppers,total=len(no_show_shoppers)):\n    rnd=random.random()\n    negative=get_negative_sample(user_idx)\n    if negative is not None:\n        if rnd<=0.2:\n            users.append(user_idx)\n            candidates.append(negative)\n            buys.append(0)\n        elif rnd<=0.3:\n            test_users.append(user_idx)\n            test_candidates.append(negative)\n            test_buys.append(0)\n\n                    \n\n        \nprint(len(users))\nprint(len(candidates))\nprint(len(buys))\n\nprint(len(test_users))\nprint(len(test_candidates))\nprint(len(test_buys))\n\n\ntraining_sample['users']=users\ntraining_sample['candidates']=candidates\ntraining_sample['buys']=buys\ntraining_sample['test_users']=test_users\ntraining_sample['test_candidates']=test_candidates\ntraining_sample['test_buys']=test_buys\ntraining_sample['shopping_history']=shopping_history\ntraining_sample['user_not_preference_dict']=user_not_preference_dict\ntraining_sample['user_preference_dict']=user_preference_dict\ntraining_sample['user_age_dict']=user_age_dict\ntraining_sample['customer_id_mapping']=customer_id_mapping\ntraining_sample['index_group_names_mapping']=index_group_names_mapping\ntraining_sample['article_id_mapping']=article_id_mapping\ntraining_sample['index_group_index_dict']=index_group_index_dict\ntraining_sample['index_group_pools']=index_group_pools\ntraining_sample['user_bought_items_dict']=user_bought_items_dict  #用戶買過的商品\n\npickle_it('./training_sample.pkl',training_sample)\nprint('pickle finished !')\n#articles_embedded_dict=unpickle('../input/h-m-recommandation/articles_embedded_dict.pkl')\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:44:15.565467Z","iopub.execute_input":"2022-09-14T14:44:15.566493Z","iopub.status.idle":"2022-09-14T14:44:48.070256Z","shell.execute_reply.started":"2022-09-14T14:44:15.566451Z","shell.execute_reply":"2022-09-14T14:44:48.069107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 節省RAM起始點\n如果你想要節省RAM，你可以從這邊開始復原你建模所需要的數據。","metadata":{}},{"cell_type":"code","source":"import shutil\n\ntraining_sample=None\nif os.path.exists('./training_sample.pkl'):\n    training_sample=unpickle('./training_sample.pkl')\n    \nelif os.path.exists('../input/h-m-recommandation-modeling/training_sample.pkl'):\n    training_sample=unpickle('../input/h-m-recommandation-modeling/training_sample.pkl')\n    shutil.copyfile('../input/h-m-recommandation-modeling/training_sample.pkl', './training_sample.pkl')\n\nusers=training_sample['users']\ncandidates=training_sample['candidates']\nbuys=training_sample['buys']\n\ntest_users=training_sample['test_users']\ntest_candidates=training_sample['test_candidates']\ntest_buys=training_sample['test_buys']\n\nuser_preference_dict=training_sample['user_preference_dict']\nuser_not_preference_dict=training_sample['user_not_preference_dict']\nshopping_history=training_sample['shopping_history']\nuser_age_dict=training_sample['user_age_dict']\n\n\ncustomer_id_mapping=training_sample['customer_id_mapping']\nindex_group_names_mapping=training_sample['index_group_names_mapping']\narticle_id_mapping=training_sample['article_id_mapping']\nindex_group_index_dict=training_sample['index_group_index_dict']\nindex_group_pools=training_sample['index_group_pools']\nuser_bought_items_dict=training_sample['user_bought_items_dict']\narticles_embedded_dict=unpickle('../input/h-m-recommandation/articles_embedded_dict.pkl')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:46:38.328958Z","iopub.execute_input":"2022-09-14T14:46:38.330216Z","iopub.status.idle":"2022-09-14T14:46:44.151240Z","shell.execute_reply.started":"2022-09-14T14:46:38.330169Z","shell.execute_reply":"2022-09-14T14:46:44.150123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"用戶=>購物歷程(時間因素與購買服飾)=>服飾(圖像特徵、文字特徵、類別特徵)","metadata":{}},{"cell_type":"code","source":"from trident.layers.pytorch_initializers import *\nclass TimeDistributed(Layer):\n    def __init__(self, module, batch_first=False):\n        super(TimeDistributed, self).__init__()\n        self.module = module\n        self.batch_first = batch_first\n\n    def forward(self, x):\n        #print('TimeDistributed',x.shape)\n        if len(x.size()) <= 2:\n            return self.module(x)\n\n        # 時間切片與批次軸合併\n        x_reshape = x.contiguous().view(-1, x.size(-1))  # (批次軸 * 時間切片, 輸入向量)\n        #print('TimeDistributed',x_reshape.shape)\n        y = self.module(x_reshape)\n        #print('TimeDistributed',y.shape)\n        # We have to reshape Y\n        if self.batch_first:\n            y = y.contiguous().view(x.size(0), -1, y.size(-1))  # (批次軸, 時間切片, 輸出向量)\n        else:\n            y = y.view(-1, x.size(1), y.size(-1))  # (時間切片, 批次軸, 輸出向量)\n        #print('TimeDistributed',y.shape)\n        return y\n    \n    \nclass Attention(Layer):\n    def __init__(self, dim=200):\n        super(Attention, self).__init__()\n        self.dim = dim\n    def build(self, input_shape: TensorShape):\n        if not self._built:\n\n            self.register_parameter('w' ,Parameter(torch.Tensor(int(input_shape[-1]),self.dim).to(get_device())))\n            self.register_parameter('b' ,Parameter(zeros( self.dim ).to(get_device())))\n            self.register_parameter('q' ,Parameter(torch.Tensor( self.dim,1).to(get_device())))\n            kaiming_uniform(self.w, a=math.sqrt(5))\n        \n            kaiming_uniform(self.q, a=math.sqrt(5))\n    def forward(self, x ,mask=None):\n        att = torch.tanh(torch.matmul(expand_dims(x,1), self.w)+ self.b)  # [batch, seq_len, hidden_dim*2]\n        att = torch.matmul(att, self.q)  # [batch, seq_len, 1]\n        att =squeeze(att, axis=1)\n\n\n        if mask is None:\n            att = exp(att)\n        else:\n            att = exp(att) * mask.float()\n\n        attention_weight = att / (reduce_sum(att, axis=-1, keepdims=True) + 1e-5)\n        #attention_weight = expand_dims(attention_weight)\n        weighted_input = x * attention_weight\n        #print('weighted_input',weighted_input.shape)\n        #print('attention_weight',attention_weight.shape)\n#         if len(weighted_input.shape)==3:\n#             weighted_input=reduce_mean(weighted_input,axis=1)\n        return weighted_input\n\n\n\n\n\nclass ArticleEncoder(Layer):\n    def __init__(self,index_group_index_dict,weight=None):\n        super(ArticleEncoder, self).__init__()\n       \n        #print(len(articles_embedded_dict),weight.shape)\n     \n        self.articlesEmbeddings=Embedding(num_embeddings=len(articles_embedded_dict),embedding_dim=1280,padding_idx=0)\n        #print(self.articlesEmbeddings.weight.shape,self.articlesEmbeddings.weight.dtype,self.articlesEmbeddings.weight.device)\n        if weight is not None:\n            self.articlesEmbeddings.weight.data.copy_(to_tensor(weight).to(get_device()))\n        self.indexgroupEmbeddings=Embedding(num_embeddings=len(index_group_names_mapping),embedding_dim=100)\n        self.index_group_index_dict=index_group_index_dict\n        \n        self.drop=Dropout(0.2)\n        self.fc=Dense(num_filters=512)\n       \n            #self.articlesEmbeddings.weight.data.copy_(to_tensor(np.stack(list(articles_embedded_dict.values()),0)))\n        self.articlesEmbeddings.trainable=False\n   \n        \n  \n    def forward(self,article_idexs):\n        if isinstance(article_idexs,list):\n            article_idexs=to_tensor(article_idexs).long().unsqueeze(-1)\n\n            \n        if ndim(article_idexs)<2:\n            article_idexs=article_idexs.unsqueeze(-1)\n    \n        #基於產品索引查表出特徵向量\n        #print('article_idexs',article_idexs.shape)\n        _device=article_idexs.device\n        #基於產品索引查表出對應的index_group_name索引\n        indexgroup_tensor= article_idexs.cpu().apply_(self.index_group_index_dict.get).to(_device)\n        \n        #to_tensor([self.index_group_dict[article_idx[0].item()] for article_idx in article_idexs]).detach().long()\n        #print('indexgroup_tensor',indexgroup_tensor.shape)\n        content_features=self.articlesEmbeddings(article_idexs).squeeze().detach()\n        \n        if ndim(content_features)<ndim(article_idexs):\n            content_features=content_features.unsqueeze(0)\n        #print('features',features.shape,features.device)\n        indexgroup_features=self.indexgroupEmbeddings(indexgroup_tensor).squeeze()\n        if ndim(indexgroup_features)<ndim(article_idexs):\n            indexgroup_features=indexgroup_features.unsqueeze(0)\n        #print('indexgroup_features',indexgroup_features.shape,indexgroup_features.device)\n        features=concate([l2_normalize(content_features,axis=-1),l2_normalize(indexgroup_features,axis=-1)],axis=-1)\n        #print('merge features',features.shape,features.device)\n        #print('features',features.shape)\n        if self.training:\n            features=self.drop(features)\n        #降維至256\n        features=self.fc(features)\n    \n        #處理attention\n        return features\n    \n    \n\nclass ShoppingEncoder(Layer):\n    def __init__(self,article_encoder):\n        super(ShoppingEncoder, self).__init__()\n        self.article_encoder=article_encoder#ArticleEncoder(weight=np.stack(articles_embedded_dict.value_list,0))\n        self.weekdayEmbeddings=Embedding(num_embeddings=7,embedding_dim=100)\n\n        self.drop=Dropout(0.2)\n        self.fc=Dense(num_filters=512)\n        self.att=Attention(512)\n \n   \n    def forward(self,article_ids):\n        \n        #分離出購買商品與星期資訊\n        weekday=article_ids[:,:,0]\n        article_ids=article_ids[:,:,1:]\n        \n        #加入article_mask，可以把padding部分遮掉，以避免非padding影響力被稀釋\n        article_mask=not_equal(article_ids,0).unsqueeze(-1).float().detach()\n        #print('article_ids.shape',weekday,weekday.shape,article_ids.shape,flush=True)\n        #基於產品索引查表出特徵向量\n        \n        article_features=self.article_encoder(article_ids)\n        #print('article_features',article_features.shape,'article_mask',article_mask.shape)\n        \n        \n        #透過加入遮罩後，再利用加總把一次購物歷程的多個商品影響疊加\n        #形狀為[批次,購物歷程數,最大購買商品數(32),通道數(256)]=>[批次,購物歷程數,通道數(256)]\n        article_features=(article_features*article_mask).sum(-2,False)\n        \n        #print('article_features',article_features.shape)\n        \n        #形狀為[批次,購物歷程數,通道數(100)]\n        weekend_features=self.weekdayEmbeddings(weekday)\n        \n        #print('article_features',article_features.shape,'weekend_features',weekend_features.shape)\n   \n        #\n        features=concate([l2_normalize(article_features,axis=-1),l2_normalize(weekend_features,axis=-1)],axis=-1)\n        #print('features(after concate)',features.shape)\n        #處理attention\n        features=self.att(features)\n        features=features.mean(-2,False)\n        \n        if self.training:\n            features=self.drop(features)\n\n        features=self.fc(features)\n        #print('features(after concate agg)',features.shape)\n        return features\n    \n    \nclass UserEncoder(Layer):\n    def __init__(self,userHistory,article_encoder,user_age_dict):\n        super(UserEncoder, self).__init__()\n        self.shopping_encoder=ShoppingEncoder(article_encoder)\n        self.userHistory=userHistory\n        self.user_age_dict=user_age_dict\n\n        self.drop=Dropout(0.2)\n        self.fc=Dense(num_filters=512)\n        self.age_fc=Dense(num_filters=512)\n        self.att=Attention(512)\n \n   \n    def forward(self,user_index):\n        \n        batch_features=[]\n        if isinstance(user_index,list):\n            user_index=to_tensor(user_index).long().unsqueeze(-1)\n            \n        user_index=user_index.long()\n        if ndim(user_index)<2:\n            user_index=user_index.unsqueeze(-1)\n            \n        #內部快取機制。\n        tmp_caches={}\n        \n        \n        all_features=[]\n        for user_idx in user_index:\n            if user_idx[0].item() in tmp_caches:\n                #如果都是相同的user就不用重複計算。\n                all_features.append(tmp_caches[user_idx[0].item()])\n            else:\n                features=self.shopping_encoder(to_tensor(self.userHistory[user_idx[0].item()]).long().unsqueeze(0))\n                age=to_tensor([self.user_age_dict[user_idx[0].item()]/100]).unsqueeze(0).unsqueeze(0)\n                #print('user age', age.shape,age.dtype,age.device)\n                age=self.age_fc(age)\n\n                features=features*age\n\n                tmp_caches[user_idx[0].item()]=features\n                all_features.append(features)\n\n        all_features=concate(all_features,0)\n        #print('user all_features', all_features.shape)\n        if self.training:\n            all_features=self.drop(all_features)\n        #降維至512\n        all_features=self.fc(all_features)\n        #print('user all_features', all_features.shape)\n     \n        return all_features","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:46:49.183839Z","iopub.execute_input":"2022-09-14T14:46:49.184316Z","iopub.status.idle":"2022-09-14T14:46:49.218164Z","shell.execute_reply.started":"2022-09-14T14:46:49.184266Z","shell.execute_reply":"2022-09-14T14:46:49.217003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 雙塔召回模型\n\n    輸入:UserIndex 以及ArticleIndex\n    輸出:這樣的人是否會購買這樣的產品(二元分類，實際上是超多品類的分類問題簡化，所以要從Softmax改為Sigmoid)\n\n如何建構用戶特徵，在本模型中是基於<font color=#FF6600>用戶(年齡與購物歷程)==>過去購物歷程(時間特徵與購買商品)==>所購買商品，根據這三個層級由底層開始彙總</font>  \n如何建構產品特徵，在本模型中是基於<font color=#FF6600>產品==>圖像Embedded (256)+ 文字Embedded (768)+ index_group_names (100)==>疊合後(1124)==>降維至(256)</font>\n","metadata":{}},{"cell_type":"markdown","source":"為了能找出效果較佳的模型，我們user embedded與article embedded融合的部分作了一點小調整，在Recommender使用的是<font color=#FF6600>批次矩陣乘法bmm</font> ，而在Recommender2所使用的是<font color=#FF6600>逐成員點乘</font> 。我們可以透過trident的學習計畫中的AB Test機制，用同樣的數據，同時訓練兩個模型，來比較這兩者何者效果叫好。在Netflex實作中是BMM遠勝過點乘，但在此案例中似乎是點乘效果較好。","metadata":{}},{"cell_type":"code","source":"import numbers\nclass Recommender(Layer):\n    def __init__(self,shopping_history):\n        super(Recommender, self).__init__()\n        self.article_encoder=ArticleEncoder(index_group_index_dict=index_group_index_dict,weight=to_tensor(np.stack(articles_embedded_dict.value_list,0).astype(np.float32)))\n        self.user_encoder =UserEncoder(shopping_history,self.article_encoder,user_age_dict)\n        \n        self.classifier =Dense(1,activation='sigmoid')\n        self.fc =Dense(64,activation='leaky_relu')\n        self.norm =L2Norm()\n        self.user_caches=OrderedDict()\n        self.article_caches=OrderedDict()\n       \n    def cache_all(self):\n        self.eval()\n        for user_index in tqdm(self.user_encoder.userHistory.keys(),total=len(self.user_encoder.userHistory.keys())):\n            user_present = l2_normalize(self.user_encoder(to_tensor([user_index]).long().unsqueeze(0)),-1)\n            #user_present=where(is_abnormal_number(user_present),zeros_like(user_present),user_present)\n            self.user_caches[user_index]=to_numpy(user_present)\n            \n        for article_index in tqdm(range(len(self.article_encoder.articlesEmbeddings.weight.data))):\n            if article_index>0:\n                article_present = l2_normalize(self.article_encoder(to_tensor([article_index]).long().unsqueeze(0)),-1)\n                #article_present=where(is_abnormal_number(article_present),zeros_like(article_present),article_present)\n                self.article_caches[article_index]=to_numpy(article_present)\n\n    def forward(self, users,candidates):\n        #print(candidates, histories)\n        #print('candidates',np.abs(to_numpy(candidates)).mean(),'histories',np.abs(to_numpy(histories)).mean())\n        user_present = l2_normalize(self.user_encoder(users),-1)\n        #user_present=where(is_abnormal_number(user_present),zeros_like(user_present),user_present)\n#         if any_abnormal_number(user_present):\n#             print('user_present nan',users.shape,users,user_present)\n\n        #print('user_present',user_present.shape)\n        article_present = l2_normalize(self.article_encoder(candidates),-1)\n        #article_present=where(is_abnormal_number(article_present),zeros_like(article_present),article_present)\n#         if any_abnormal_number(article_present):\n#             print('article_present nan',candidates.shape,candidates,article_present)\n        #print('article_present',article_present.shape)\n    \n        preds=torch.bmm(article_present.unsqueeze(-1), user_present.unsqueeze(1)).mean(1)\n#         if any_abnormal_number(preds):\n#             print('preds nan')\n        #print('preds_bmm',np.abs(to_numpy(preds)).mean())\n        preds=self.fc(preds)\n        preds=self.norm(preds)\n        #print('preds',np.abs(to_numpy(preds)).mean())\n        preds=self.classifier(preds)\n        #print('preds',np.abs(to_numpy(preds)).mean())\n        return preds\n    \n    def fast_forward(self,user,article):\n        if hasattr(user, '__iter__') and hasattr(article, '__iter__'):\n            return self.batch_forward(user,article)\n        elif isinstance(user,numbers.Integral) and isinstance(article,numbers.Integral):\n            return self.singleton_forward(user,article)\n        \n    def singleton_forward(self, user_index,candidate_index):\n        user_present=None\n        if user_index in self.user_caches:\n            user_present = to_tensor(self.user_caches[user_index])\n        else:\n            user_present = l2_normalize(self.user_encoder(to_tensor([user_index]).long().unsqueeze(0)),-1)\n#         if any_abnormal_number(user_present):\n#             print('user_present nan',users.shape,users,user_present)\n\n        article_present=None\n        if candidate_index in self.article_caches:\n            article_present = to_tensor(self.article_caches[candidate_index])\n        else:\n            article_present = l2_normalize(self.article_encoder(to_tensor([candidate_index]).long().unsqueeze(0)),-1)\n       \n        preds=torch.bmm(article_present.unsqueeze(-1), user_present.unsqueeze(1)).mean(1)\n\n        preds=self.fc(preds)\n        #preds=self.norm(preds)\n        preds=self.classifier(preds)\n        return preds\n    \n    def batch_forward(self, user_indexs,candidate_indexs):\n        results=[]\n        for user_idx,article_idx in zip(user_indexs,candidate_indexs):\n            results.append(to_numpy(self.singleton_forward(user_idx,article_idx))[0].item())\n        return results\n    \nclass Recommender2(Layer):\n    def __init__(self,shopping_history):\n        super(Recommender2, self).__init__()\n        self.article_encoder=ArticleEncoder(index_group_index_dict=index_group_index_dict,weight=to_tensor(np.stack(articles_embedded_dict.value_list,0).astype(np.float32)))\n        self.user_encoder =UserEncoder(shopping_history,self.article_encoder,user_age_dict)\n        \n        self.classifier =Dense(1,activation='sigmoid')\n        self.fc =Dense(64,activation='leaky_relu')\n        self.norm =L2Norm()\n        self.user_caches=OrderedDict()\n        self.article_caches=OrderedDict()\n       \n    def cache_all(self):\n        self.eval()\n        for user_index in tqdm(self.user_encoder.userHistory.keys(),total=len(self.user_encoder.userHistory.keys())):\n            user_present = l2_normalize(self.user_encoder(to_tensor([user_index]).long().unsqueeze(0)),-1)\n            \n            self.user_caches[user_index]=to_numpy(user_present)\n            \n        for article_index in tqdm(range(len(self.article_encoder.articlesEmbeddings.weight.data))):\n            if article_index>0:\n                article_present = l2_normalize(self.article_encoder(to_tensor([article_index]).long().unsqueeze(0)),-1)\n                \n                self.article_caches[article_index]=to_numpy(article_present)\n\n    def forward(self, users,candidates):\n        user_present = l2_normalize(self.user_encoder(users),-1)\n\n        article_present = l2_normalize(self.article_encoder(candidates),-1)\n\n        preds=article_present*user_present\n\n        preds=self.fc(preds)\n        preds=self.norm(preds)\n \n        preds=self.classifier(preds)\n     \n        return preds\n    \n    def fast_forward(self,user,article):\n        if hasattr(user, '__iter__') and hasattr(article, '__iter__'):\n            return self.batch_forward(user,article)\n        elif isinstance(user,numbers.Integral) and isinstance(article,numbers.Integral):\n            return self.singleton_forward(user,article)\n        else:\n            print(user,article)\n        \n    def singleton_forward(self, user_index,candidate_index):\n        user_present=None\n        if user_index in self.user_caches:\n            user_present = to_tensor(self.user_caches[user_index])\n        else:\n            user_present = l2_normalize(self.user_encoder(to_tensor([user_index]).long().unsqueeze(0)),-1)\n#         if any_abnormal_number(user_present):\n#             print('user_present nan',users.shape,users,user_present)\n\n        article_present=None\n        if candidate_index in self.article_caches:\n            article_present = to_tensor(self.article_caches[candidate_index])\n        else:\n            article_present = l2_normalize(self.article_encoder(to_tensor([candidate_index]).long().unsqueeze(0)),-1)\n       \n        preds=article_present*user_present\n\n        preds=self.fc(preds)\n        preds=self.norm(preds)\n        preds=self.classifier(preds)\n        return preds\n    \n    def batch_forward(self, user_indexs,candidate_indexs):\n        results=[]\n        for user_idx,article_idx in zip(user_indexs,candidate_indexs):\n            results.append(to_numpy(self.singleton_forward(user_idx,article_idx))[0].item())\n        return results","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:46:49.303686Z","iopub.execute_input":"2022-09-14T14:46:49.304563Z","iopub.status.idle":"2022-09-14T14:46:49.333947Z","shell.execute_reply.started":"2022-09-14T14:46:49.304516Z","shell.execute_reply":"2022-09-14T14:46:49.332779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(candidates[:10])\nprint(users[:10])\nprint(buys[:10])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:46:49.337087Z","iopub.execute_input":"2022-09-14T14:46:49.337667Z","iopub.status.idle":"2022-09-14T14:46:49.351757Z","shell.execute_reply.started":"2022-09-14T14:46:49.337629Z","shell.execute_reply":"2022-09-14T14:46:49.350581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 建構數據提供者\n","metadata":{}},{"cell_type":"code","source":"ds1=NumpyDataset(np.expand_dims(np.array(candidates),-1).astype(np.int64),symbol='candidates')\nds2=NumpyDataset(np.expand_dims(np.array(users),-1).astype(np.int64),symbol='users')\nds3=NumpyDataset(np.expand_dims(np.array(buys),-1).astype(np.float32),symbol='buys')\n\nds1_test=NumpyDataset(np.expand_dims(np.array(test_candidates),-1).astype(np.int64),symbol='candidates')\nds2_test=NumpyDataset(np.expand_dims(np.array(test_users),-1).astype(np.int64),symbol='users')\nds3_test=NumpyDataset(np.expand_dims(np.array(test_buys),-1).astype(np.float32),symbol='buys')\n\n\nzipdataset=ZipDataset(ds1,ds2)\nzipdataset_test=ZipDataset(ds1_test,ds2_test)\ndata_provider=DataProvider(traindata=Iterator(data=zipdataset,label=ds3),testdata=Iterator(data=zipdataset_test,label=ds3_test))\nprint(data_provider.signature)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:46:49.353484Z","iopub.execute_input":"2022-09-14T14:46:49.354480Z","iopub.status.idle":"2022-09-14T14:46:54.558045Z","shell.execute_reply.started":"2022-09-14T14:46:49.354439Z","shell.execute_reply":"2022-09-14T14:46:54.556119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_candidates,_users,_buys=data_provider.next()\nprint(_candidates.shape,_users.shape,_buys.shape)\nprint(_candidates)\nprint(_buys)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:46:54.559585Z","iopub.execute_input":"2022-09-14T14:46:54.560306Z","iopub.status.idle":"2022-09-14T14:46:54.572649Z","shell.execute_reply.started":"2022-09-14T14:46:54.560265Z","shell.execute_reply":"2022-09-14T14:46:54.571485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"net1是矩陣乘法，net2是使用點乘，在這個案例中實測後者效果比較好。此外，ShoppingEncoder中也加入了mask遮蔽掉padding的部分。同時UserEncoder也加入了內部快取的機制。","metadata":{}},{"cell_type":"code","source":"articles_embedded_dict=unpickle('../input/h-m-recommandation/articles_embedded_dict.pkl')\nfor k in articles_embedded_dict.key_list:\n    if articles_embedded_dict[k].shape==(1024,):\n        articles_embedded_dict[k]=np.concatenate([np.random.uniform(-0.02,0.02,512),articles_embedded_dict[k][256:]])\npickle_it('articles_embedded_dict.pkl',articles_embedded_dict)\n\n\nprint(list(set([m.shape for m in articles_embedded_dict.value_list])))\n#net1=Model(inputs=(_users,_candidates),output=Recommender(shopping_history))\nnet2=Model(inputs=(_users,_candidates),output=Recommender2(shopping_history))\n#權重初始化\n#xavier_normal(net1.model,0.02)\n#net1.summary()\n\nxavier_normal(net2.model,0.02)\n#net2.load_model('./Models/net2.pth.tar')\nnet2.summary()\n\ndel articles_embedded_dict\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:52:51.575624Z","iopub.execute_input":"2022-09-14T14:52:51.576341Z","iopub.status.idle":"2022-09-14T14:52:55.655087Z","shell.execute_reply.started":"2022-09-14T14:52:51.576299Z","shell.execute_reply":"2022-09-14T14:52:55.654037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 多目標優化  \n\n* bce_focal_loss希望能從稀有事件中精確地找出購買事件  \n* f1_score在兼顧準確率的同時也要確保召回率  \n* punishment 如果模型產出接近50%機率的曖昧不明答案則給予懲罰  \n* positive_ratio 希望最後產出的名單越少越好  \n\n總之就是就是希望能夠生成精確度高、召回率高、答案明確的名單。\n","metadata":{}},{"cell_type":"code","source":"#把focal loss加入至binary cross entropy loss中\ndef bce_focal_loss(output,buys):\n    B=output.size(0)\n    #print(output)\n    buys=buys.detach().reshape(-1)\n#     gamma=2.5\n#     pos_weight=5\n    #output=where(is_abnormal_number(output),zeros_like(output),output)\n    output = clip(output, 1e-7, 1 - 1e-7).reshape(-1)\n   \n    loss_weight=where(buys==1,3.5*ones_like(buys),ones_like(buys))\n    \n\n    bce = loss_weight*buys * torch.log(output)#* ((1 - output) ** gamma)\n\n    bce +=(1 - buys) * torch.log(1 - output)#* ((output) ** gamma)\n   \n    #print('bce2',bce)\n    #return (loss_weight*binary_cross_entropy(output,buys,True)).sum()/B\n    return -bce.sum()/B\n   \n\n    \n    #return binary_cross_entropy(output,buys,True).mean()\n    \n#平衡準確度與召回度\ndef f1_score(output,buys):\n    beta=1.2#比1大則重recall\n    output = clip(output, 1e-7, 1 - 1e-7).squeeze()\n    buys=buys.squeeze(-1).float().detach()\n    tp = reduce_sum(buys * output)\n \n    precision = tp / clip(reduce_sum(output),min=1)\n    recall = tp / clip(reduce_sum(buys),min=1)\n    return 1 - (1 + beta ** 2) * precision * recall / (beta ** 2 * precision + recall)\n\n#若是模型都產出接近0.5這類曖昧不明的預測則會遭到懲罰，明確的1,0懲罰會最小\ndef punishment(output,buys):\n    mask=buys==1\n    return 0.5*(1-(reduce_mean(output[mask])-reduce_mean(output[~mask])))\n    \ndef accuracy1(output,buys):\n#     print('output',output.shape,output)\n#     print('buys',buys.shape,buys)\n    \n    output = clip(output, 1e-7, 1 - 1e-7).squeeze()\n    buys=buys.squeeze(-1).float().detach()\n    return (greater(output,0.5)*buys).sum()/clip(greater(output,0.5).sum(),min=1)\n\ndef recall1(output,buys):\n    output = clip(output, 1e-7, 1 - 1e-7).squeeze()\n    buys=buys.squeeze(-1).float().detach()\n    return (greater(output,0.5)*buys).sum()/clip(buys.sum(),min=1)\n\ndef positive_ratio(output):\n    return greater(output,0.5).float().mean()\n\ndef output_mean(output):\n    return output.mean()\n\n\n#除了傳入抽樣組成的正樣本與負樣本，會造成訓練的龜速以及不穩定。因此我們利用了服飾的性別特質來定義一整包的負樣本(對只買男裝的人來說，所有女裝都是負樣本)\n#同時將所有過去買過的商品變成效果減弱的正樣本\n#這樣可以加速訓練\ndef not_prefer_loss(users,output,buys):\n    mask=buys==1\n    buy_users=users[mask]\n    buy_output=output[mask]\n    loss=0\n    \n    for i in range(len(users)):\n        user_idx=users[i].item()\n        #print(user_idx.shape,user_idx)\n        \n        if user_idx in user_not_preference_dict:\n            not_prefer=list(set(user_not_preference_dict[user_idx]))\n            #處理性別特質明確\n            #只買女裝從未買男裝者，以男裝作為他的負樣本\n            items=None\n            if 'Menswear' in not_prefer and 'Ladieswear' not in not_prefer:\n                items=index_group_pools['Menswear']\n            #只買男裝從未買女裝者，以女裝作為他的負樣本\n            elif 'Ladieswear' in not_prefer and 'Menswear' not in not_prefer:\n                items=index_group_pools['Ladieswear']\n            elif len(not_prefer)==1 and 'Baby/Children'!=not_prefer[0]:\n                items=index_group_pools[not_prefer[0]]\n                \n            if 'Baby/Children' in not_prefer:\n                baby_items=index_group_pools['Baby/Children']\n                random.shuffle(baby_items)\n                if items is None:\n                    items=baby_items[:32]\n                else:\n                    items.extend(baby_items[:32])\n                \n            if items is not None:\n                if len(items)>64:\n                    random.shuffle(items)\n                    items=items[:64]\n                items=list(set(items))\n                result=net2(to_tensor([user_idx]*len(items)).long().unsqueeze(-1),to_tensor(items).long().unsqueeze(-1))\n                loss=loss+binary_cross_entropy(result,zeros_like(result),True).mean()\n                #歷史曾買過之商品做為正樣本\n        if user_idx in user_bought_items_dict:\n            bought_items=list(set(user_bought_items_dict[user_idx]))\n            if len(bought_items)>16:\n                random.shuffle(bought_items)\n                bought_items=bought_items[:len(bought_items)//2]\n            result2=net2(to_tensor([user_idx]*len(bought_items)).long().unsqueeze(-1),to_tensor(bought_items).long().unsqueeze(-1))\n            loss=loss+binary_cross_entropy(result2,ones_like(result2),True).mean()\n        #loss=loss+binary_cross_entropy(buy_output[i],ones_like(buy_output[i]),True).mean()\n\n    return loss\n\n\ndef article_match_loss(users):\n    mask=buys==1\n\n    loss=0\n    n=1\n    for i in range(len(users)):\n        user_idx=users[i].item()\n        user_present =net2.model.user_encoder(users[i].unsqueeze(-1))\n\n        #print(user_idx.shape,user_idx)\n        \n        if user_idx in user_not_preference_dict:\n            not_prefer=list(set(user_not_preference_dict[user_idx]))\n            #處理性別特質明確\n            #只買女裝從未買男裝者，以男裝作為他的負樣本\n            items=None\n            if 'Menswear' in not_prefer and 'Ladieswear' not in not_prefer:\n                items=index_group_pools['Menswear']\n            #只買男裝從未買女裝者，以女裝作為他的負樣本\n            elif 'Ladieswear' in not_prefer and 'Menswear' not in not_prefer:\n                items=index_group_pools['Ladieswear']\n            elif len(not_prefer)==1 and 'Baby/Children'!=not_prefer[0]:\n                items=index_group_pools[not_prefer[0]]\n                \n            if 'Baby/Children' in not_prefer:\n                baby_items=index_group_pools['Baby/Children']\n                random.shuffle(baby_items)\n                if items is None:\n                    items=baby_items[:32]\n                else:\n                    items.extend(baby_items[:32])\n                \n            if items is not None and user_idx in user_bought_items_dict:\n                items=list(set(items))\n                if len(items)>16:\n                    random.shuffle(items)\n                    items=items[:16]\n                \n                bought_items=list(set(user_bought_items_dict[user_idx]))\n                if len(bought_items)>16:\n                    random.shuffle(bought_items)\n                    bought_items=bought_items[:16]\n                #根據不會去購買的品類所建構的負樣本表徵\n                negative_embedded= net2.model.article_encoder(to_tensor(items).long().unsqueeze(-1))\n                #根據過去曾購買的品類所建構的正樣本表徵\n                positive_embedded=net2.model.article_encoder(to_tensor(bought_items).long().unsqueeze(-1))\n                \n                \n                \n                #print(negative_embedded.shape,positive_embedded.shape,user_present.shape)\n                \n                #((negative_embedded*user_present).mean()-0)**2+((positive_embedded*user_present).mean()-1)**2\n                #使用者表徵跟正樣本表徵的cosine距離應該要比跟負樣本來的更大(cosine距離是值越大越像)\n                loss=loss+((element_cosine_distance(user_present,negative_embedded,axis=-1).mean()+0)**2+(element_cosine_distance(user_present,positive_embedded,axis=-1).mean()-1)**2)\n                n+=1\n    \n    return loss/n\n        \n\nfrom matplotlib import pyplot as plt \n\ntransactions_p1m=unpickle('../input/h-m-recommandation-modeling/transactions_p1m.pkl')\n#每隔100批次，隨機挑一個用戶，計算它在每個產品的輸出機率分布，如果都集中在某處則不是理想的結果。\ndef try_recall(training_context):\n    if training_context['current_batch']>0 and ((training_context['current_batch']+1)%200==0 ):\n        if training_context['current_batch']>=6000:\n            training_context['stop_training']=True\n            return None\n        model=training_context['current_model']\n        model.eval()\n        user_index=random.choice(list(model.user_encoder.userHistory.keys()))\n        print('Try Recall UserIndex {0}'.format(user_index))\n\n        item=[article_id_mapping[article_id] for article_id in transactions_p1m['article_id'].unique()]\n        user = [user_index] * len(item)\n        results=to_numpy(model.forward(user, item))\n        bins = [0,0.1,0.2,0.3,0.4,0.5,0.6,0.7,0.8,0.9,1]\n        hist,_bins = np.histogram(results,bins = [0,0.1,0.2,0.3,0.4,0.5,0.6,0.7,0.8,0.9,1]) \n        print(bins)\n        print(hist)\n        plt.hist(results, bins = [0,0.1,0.2,0.3,0.4,0.5,0.6,0.7,0.8,0.9,1]) \n        plt.title(\"histogram\") \n        plt.axis(\"off\")\n        display.display(plt.gcf())\n        plt.close()\n\n\n        \n\n\n\n# net1.with_optimizer(optimizer=Adam,lr=5e-4,betas=(0.9, 0.999),gradient_centralization='all')\\\n#     .with_loss(bce_focal_loss,loss_weight=16)\\\n#     .with_loss(f1_score,loss_weight=8)\\\n#     .with_loss(punishment)\\\n#     .with_loss(positive_ratio,name='positive_ratio',as_metric=True)\\\n#     .with_metric(accuracy1,name='accuracy')\\\n#     .with_metric(recall1,name='recall')\\\n#     .with_metric(output_mean,print_only=True,name='output_mean')\\\n#     .with_regularizer('l2',reg_weight=1e-5)\\\n#     .with_learning_rate_scheduler(CosineLR(period=3000,unit='batch', min_lr=1e-5))\\\n#     .trigger_when(frequency=100,action=try_recall)\\\n#     .with_model_save_path('Models/net1.pth')\\\n\n\nnet2.with_optimizer(optimizer=DiffGrad,lr=5e-5,betas=(0.9, 0.999),gradient_centralization='all')\\\n    .with_loss(bce_focal_loss,loss_weight=16)\\\n    .with_loss(f1_score,loss_weight=6)\\\n    .with_loss(article_match_loss,loss_weight=2)\\\n    .with_loss(punishment,loss_weight=0.2)\\\n    .with_loss(positive_ratio,loss_weight=1,name='positive_ratio',as_metric=True)\\\n    .with_metric(accuracy1,name='accuracy')\\\n    .with_metric(recall1,name='recall')\\\n    .with_metric(output_mean,print_only=True,name='output_mean')\\\n    .with_regularizer('l2',reg_weight=1e-5)\\\n    .with_grad_clipping(3)\\\n    .with_learning_rate_scheduler(CosineLR(period=3000,unit='batch', min_lr=1e-5))\\\n    .trigger_when(frequency=10,action=try_recall)\\\n    .with_model_save_path('Models/net2.pth')\\\n   # .with_automatic_mixed_precision_training()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:53:02.724702Z","iopub.execute_input":"2022-09-14T14:53:02.725588Z","iopub.status.idle":"2022-09-14T14:53:03.631254Z","shell.execute_reply.started":"2022-09-14T14:53:02.725545Z","shell.execute_reply":"2022-09-14T14:53:03.630067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plan=TrainingPlan()\\\n    .add_training_item(net2)\\\n    .with_data_loader(data_provider)\\\n    .repeat_epochs(3)\\\n    .with_batch_size(128)\\\n    .print_progress_scheduling(10,unit='batch')\\\n    .display_loss_metric_curve_scheduling(frequency=200,unit='batch',imshow=True)\\\n    .out_sample_evaluation_scheduling(frequency=100,unit='batch')\\\n    .save_model_scheduling(50,unit='batch')\\","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:53:03.633758Z","iopub.execute_input":"2022-09-14T14:53:03.634186Z","iopub.status.idle":"2022-09-14T14:53:03.642044Z","shell.execute_reply.started":"2022-09-14T14:53:03.634142Z","shell.execute_reply":"2022-09-14T14:53:03.641005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plan.start_now()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:53:03.643678Z","iopub.execute_input":"2022-09-14T14:53:03.644371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"召回模型初步訓練好了後，那要如何進行召回呢？如果一次送入一個客戶索引以及一個產品索引來進行評分，雖然這是正確的步驟，但是鐵定非常龜速。那該如何提升效率呢?還記得課程中說過，<font color=#6a5acd>在工業上為何喜歡雙塔模型，其中的一個原因是在短期間內(以服飾業來說，一周內都還算變動不大)，可以把產品以及客戶的Embedded視為固定值。那我們就可以把它整批算好儲存起來，要用的時候再調用，</font>這樣就可以省下不少計算量。","metadata":{}},{"cell_type":"code","source":"\nnet2.model.eval()\nnet2.model.user_caches=OrderedDict()\nnet2.model.article_caches=OrderedDict()\nnet2.model.cache_all()\n\npickle_it('./user_caches.pkl',net2.model.user_caches)\npickle_it('./article_caches.pkl',net2.model.article_caches)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}