{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%matplotlib inline\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom IPython import display\nplt.rcParams.update({'figure.figsize': [10,10]})","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-14T11:21:25.586224Z","iopub.execute_input":"2022-09-14T11:21:25.592660Z","iopub.status.idle":"2022-09-14T11:21:25.678367Z","shell.execute_reply.started":"2022-09-14T11:21:25.592543Z","shell.execute_reply":"2022-09-14T11:21:25.677022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# H&M 個人化時尚產品推薦\n\n時尚產品的推薦:\n* 性別差異以及中性服飾\n* 嬰兒服飾推薦\n* 季節性商品\n* 會重複購買語不會重複購買的商品型態\n","metadata":{}},{"cell_type":"code","source":"import glob\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nos.environ['TRIDENT_BACKEND'] = 'pytorch'\n!pip uninstall tridentx -y\n!pip install ../input/trident/tridentx-0.7.5-py3-none-any.whl --upgrade\nimport trident as T\nfrom trident import *","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:21:25.681266Z","iopub.execute_input":"2022-09-14T11:21:25.682210Z","iopub.status.idle":"2022-09-14T11:21:53.169204Z","shell.execute_reply.started":"2022-09-14T11:21:25.682152Z","shell.execute_reply":"2022-09-14T11:21:53.167808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_articles=pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_articles['text']=df_articles['index_name']+', '+df_articles['section_name']+', Product type:'+df_articles['product_type_name']+', '+df_articles['detail_desc']+', Product:'+df_articles['prod_name']+', Color:'+df_articles['colour_group_name']+', Graphical appearance:'+df_articles['graphical_appearance_name']\ndf_articles.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:21:53.171457Z","iopub.execute_input":"2022-09-14T11:21:53.172498Z","iopub.status.idle":"2022-09-14T11:21:54.980078Z","shell.execute_reply.started":"2022-09-14T11:21:53.172453Z","shell.execute_reply":"2022-09-14T11:21:54.978877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_ids=df_articles['article_id'].unique()\narticle_id_mapping={}\narticle_id_mapping[0]=0\narticle_id_mapping = {id:i+1 for i, id in enumerate(article_ids)}","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:21:54.984518Z","iopub.execute_input":"2022-09-14T11:21:54.985970Z","iopub.status.idle":"2022-09-14T11:21:55.035598Z","shell.execute_reply.started":"2022-09-14T11:21:54.985924Z","shell.execute_reply":"2022-09-14T11:21:55.034235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keys=df_articles['article_id'].to_numpy()\n\nvalues=df_articles['text'].to_numpy().astype(np.str)\nprint(values[:5],values.dtype)\nprint(values[0].item())\ntext_dict=OrderedDict(zip(keys,values))\nprint(text_dict.value_list[0])","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:21:55.039752Z","iopub.execute_input":"2022-09-14T11:21:55.040092Z","iopub.status.idle":"2022-09-14T11:21:55.829603Z","shell.execute_reply.started":"2022-09-14T11:21:55.040062Z","shell.execute_reply":"2022-09-14T11:21:55.827856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer,BertModel,BertTokenizerFast\ntokenizer = BertTokenizer.from_pretrained(\"bert-base-uncased\")\nbert_model = BertModel.from_pretrained('bert-base-uncased').cuda()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:21:55.833906Z","iopub.execute_input":"2022-09-14T11:21:55.835343Z","iopub.status.idle":"2022-09-14T11:22:49.632071Z","shell.execute_reply.started":"2022-09-14T11:21:55.835295Z","shell.execute_reply":"2022-09-14T11:22:49.630556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder=load('../input/h-m-articles-similarity/Models/articles_embedded_seresnt.pth')\n\nencoder=Model(input_shape=(3,128,128),output=encoder[0])\nencoder.summary()","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:22:49.634831Z","iopub.execute_input":"2022-09-14T11:22:49.635230Z","iopub.status.idle":"2022-09-14T11:23:02.353434Z","shell.execute_reply.started":"2022-09-14T11:22:49.635198Z","shell.execute_reply":"2022-09-14T11:23:02.351553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_embedded_dict=OrderedDict()\nfor i in range(len(article_id_mapping)):\n    articles_embedded_dict[i]=None\n\nresize_fn=Resize((128,128),True,background_color=(236,235,233))    \n\ndef prepare_image(img_path):\n    img=image2array(img_path)\n    if len(img.shape)==2:\n        img=np.stack([img,img,img],axis=-1) \n    img=img[10:-10,10:-10,:]\n    img=resize_fn(img)\n    img=(img-127.5)/127.5\n    img=to_tensor(image_backend_adaption(img)).unsqueeze(0)\n    return img\n\n\nimgs=list(sorted(set(glob.glob('../input/h-and-m-personalized-fashion-recommendations/images/*/*.*g'))))\n\n\n\narticles_embedded_dict[0]=np.random.uniform(-0.02,0.02,1024)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:23:58.732593Z","iopub.execute_input":"2022-09-14T11:23:58.734013Z","iopub.status.idle":"2022-09-14T11:23:59.301941Z","shell.execute_reply.started":"2022-09-14T11:23:58.733926Z","shell.execute_reply":"2022-09-14T11:23:59.300478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nbatch_ids=[]\nbatch_images=[]\nbatch_texts=[]\n\nfor img_path in tqdm(imgs):\n    try:\n        folder,filename,ext=split_path(img_path)\n        filename=int(filename)\n        if articles_embedded_dict[article_id_mapping[filename]] is None:\n            inp=prepare_image(img_path)\n            batch_ids.append(filename)\n            batch_images.append(inp)\n            batch_texts.append(str(text_dict[filename]))\n\n            #每32筆一個批次，或是到達最後一筆時執行批次推論\n            if len(batch_ids)==32 or img_path==imgs[-1]:\n\n                #產生圖像表徵，長度256\n                batch_images=concate(batch_images,axis=0)\n                image_embeddings=to_numpy(encoder(batch_images))\n\n\n                encoding = tokenizer.batch_encode_plus(\n                      batch_text_or_text_pairs=batch_texts,\n                      max_length=256,           # max length of sentence \n                      add_special_tokens=True, # Add '[CLS]' and '[SEP]'\n                      return_token_type_ids=False,\n                      padding='max_length',\n                      truncation=True,\n                      return_attention_mask=True,\n                      return_tensors='pt',  # Return PyTorch tensors\n                    )\n\n                output = bert_model(\n                  input_ids=encoding['input_ids'].to(get_device()),\n                  attention_mask=encoding['attention_mask'].to(get_device())\n                )\n                #產生文字描述表徵，長度768\n                text_embeddings=to_numpy(output['pooler_output'])\n                for k in range(len(batch_ids)):\n                    articles_embedded_dict[article_id_mapping[batch_ids[k]]]=np.concatenate([image_embeddings[k],l2_normalize(text_embeddings[k])])\n                batch_ids=[]\n                batch_images=[]\n                batch_texts=[]\n            \n    except Exception as e:\n        \n        print(e)\n#         PrintException()\n#         break\n\nprint(len(articles_embedded_dict))","metadata":{"execution":{"iopub.status.busy":"2022-09-14T11:23:59.304755Z","iopub.execute_input":"2022-09-14T11:23:59.305285Z","iopub.status.idle":"2022-09-14T14:16:30.606837Z","shell.execute_reply.started":"2022-09-14T11:23:59.305226Z","shell.execute_reply":"2022-09-14T14:16:30.604250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#有文字沒圖片\nfor k,v in text_dict.items():\n    idx=article_id_mapping[k]\n    if idx in articles_embedded_dict and articles_embedded_dict[idx] is None:\n        encoding = tokenizer.encode_plus(\n                  text=v,\n                  max_length=256,           # max length of sentence \n                  add_special_tokens=True, # Add '[CLS]' and '[SEP]'\n                  return_token_type_ids=False,\n                  padding='max_length',\n                  truncation=True,\n                  return_attention_mask=True,\n                  return_tensors='pt',  # Return PyTorch tensors\n                )\n            \n        output = bert_model(\n          input_ids=encoding['input_ids'].to(get_device()),\n          attention_mask=encoding['attention_mask'].to(get_device())\n        )\n        #產生文字描述表徵，長度768\n        text_embedding=to_numpy(output['pooler_output'][0])\n        articles_embedded_dict[article_id_mapping[idx]]=np.concatenate([np.random.uniform(-0.02,0.02,256),text_embedding])\n#確認一遍有無疏漏     \nfor k,v in articles_embedded_dict.items():\n    if v is None:\n        print(k)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:26:19.679597Z","iopub.execute_input":"2022-09-14T14:26:19.680249Z","iopub.status.idle":"2022-09-14T14:26:19.993729Z","shell.execute_reply.started":"2022-09-14T14:26:19.680191Z","shell.execute_reply":"2022-09-14T14:26:19.992297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(imgs))\nprint(len(articles_embedded_dict))","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:26:25.953665Z","iopub.execute_input":"2022-09-14T14:26:25.954126Z","iopub.status.idle":"2022-09-14T14:26:25.962511Z","shell.execute_reply.started":"2022-09-14T14:26:25.954092Z","shell.execute_reply":"2022-09-14T14:26:25.961085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickle_it('./articles_embedded_dict.pkl',articles_embedded_dict)","metadata":{"execution":{"iopub.status.busy":"2022-09-14T14:26:45.638511Z","iopub.execute_input":"2022-09-14T14:26:45.638956Z","iopub.status.idle":"2022-09-14T14:26:48.583866Z","shell.execute_reply.started":"2022-09-14T14:26:45.638908Z","shell.execute_reply":"2022-09-14T14:26:48.581630Z"},"trusted":true},"execution_count":null,"outputs":[]}]}