{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nimport networkx as nx\nfrom gensim.models import Word2Vec\nfrom sklearn.metrics.pairwise import cosine_similarity\nimport matplotlib.image as mpimg\nimport random\nfrom tqdm import tqdm\ntqdm.pandas()\nfrom collections import defaultdict, Counter","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-11T01:49:04.611068Z","iopub.execute_input":"2023-07-11T01:49:04.611545Z","iopub.status.idle":"2023-07-11T01:49:04.619926Z","shell.execute_reply.started":"2023-07-11T01:49:04.611506Z","shell.execute_reply":"2023-07-11T01:49:04.618873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:49:04.622547Z","iopub.execute_input":"2023-07-11T01:49:04.623134Z","iopub.status.idle":"2023-07-11T01:49:35.828471Z","shell.execute_reply.started":"2023-07-11T01:49:04.623110Z","shell.execute_reply":"2023-07-11T01:49:35.827044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_freq = transactions.groupby('article_id')['customer_id'].nunique()\nuser_freq = transactions.groupby('customer_id')['article_id'].nunique()\n\nitems = item_freq[item_freq >= 100].index\nusers = user_freq[user_freq >= 100].index\n\nfiltered_df = transactions[transactions['article_id'].isin(items) & transactions['customer_id'].isin(users)]","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:49:35.833763Z","iopub.execute_input":"2023-07-11T01:49:35.835249Z","iopub.status.idle":"2023-07-11T01:50:26.528187Z","shell.execute_reply.started":"2023-07-11T01:49:35.835182Z","shell.execute_reply":"2023-07-11T01:50:26.526617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freq = filtered_df.groupby(['customer_id', 'article_id']).size().reset_index(name='frequency')\n\nGraphTravel_HM = filtered_df.merge(freq, on=['customer_id', 'article_id'], how='left')\n\nGraphTravel_HM = GraphTravel_HM[GraphTravel_HM['frequency'] >= 10]\n\nGraphTravel_HM","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:50:26.533107Z","iopub.execute_input":"2023-07-11T01:50:26.533429Z","iopub.status.idle":"2023-07-11T01:50:36.160534Z","shell.execute_reply.started":"2023-07-11T01:50:26.533407Z","shell.execute_reply":"2023-07-11T01:50:36.159667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"G = nx.Graph()\n\nfor index, row in GraphTravel_HM.iterrows():\n    G.add_node(row['customer_id'], type='user')\n    G.add_node(row['article_id'], type='item')\n    G.add_edge(row['customer_id'], row['article_id'], weight=row['frequency'])","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:50:36.161474Z","iopub.execute_input":"2023-07-11T01:50:36.161853Z","iopub.status.idle":"2023-07-11T01:50:37.462241Z","shell.execute_reply.started":"2023-07-11T01:50:36.161831Z","shell.execute_reply":"2023-07-11T01:50:37.461312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# G.nodes['06636eaa476a2f3417e8d11905e17a8066b4c9ae26bc785eb8a55e0b8ba29d9e']","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:50:37.463316Z","iopub.execute_input":"2023-07-11T01:50:37.464843Z","iopub.status.idle":"2023-07-11T01:50:37.470972Z","shell.execute_reply.started":"2023-07-11T01:50:37.464789Z","shell.execute_reply":"2023-07-11T01:50:37.469386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#randomwalkを実行する関数を定義\nimport networkx as nx\n\ndef make_random_walks(G, num_walk, length_of_walk):\n    #ランダムウォークで歩いたノードを入れるlistを生成\n    paths = list()\n    #ランダムウォークを擬似的に行う\n    for i in tqdm(range(num_walk)):\n        node_list = list(G.nodes())\n        for node in node_list:\n            now_node = node\n            #到達したノードを追加する用のリストを用意する\n            path = list()\n            path.append(str(now_node))\n            for j in range(length_of_walk):\n                #次に到達するノードを選択する\n                next_node = random.choice(list(G.neighbors(now_node)))\n                #リストに到達したノードをリストに追加する\n                path.append(str(next_node))\n                #今いるノードを「現在地」とする\n                now_node = next_node\n            #ランダムウォークしたノードをリストに追加\n            paths.append(path)\n        #訪れたノード群を返す\n        return paths\n","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:50:37.473348Z","iopub.execute_input":"2023-07-11T01:50:37.473768Z","iopub.status.idle":"2023-07-11T01:50:37.490580Z","shell.execute_reply.started":"2023-07-11T01:50:37.473739Z","shell.execute_reply":"2023-07-11T01:50:37.488313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ランダムウォークで取得したノードをWord2Vecに入力\n\nfrom gensim.models import Word2Vec as word2vec\n\nwalking = make_random_walks(G, 512, 512)\nmodel = word2vec(walking, min_count = 1, vector_size = 2, window = 10, workers = 4)\n\nx = list()\ny = list()\nnode_list = list()\ncolors = list()\nfig, ax = plt.subplots()\nfor node in G.nodes:\n    #int型のままではイテレートできないので、string型に変換する\n    vector = model.wv[str(node)]  \n    x.append(vector[0])\n    y.append(vector[1])\n    #注釈として、ノードの番号を追記する\n    #座標(x,y)は(vector[0],vector[1])を指定\n#     ax.annotate(str(node), (vector[0], vector[1]))\n#     if G.nodes[node][\"club\"] == \"Officer\":\n#         colors.append(\"r\")\n#     else:\n#         colors.append(\"b\")\nfor i in range(len(x)):\n#     ax.scatter(x[i], y[i], c=colors[i])\n    ax.scatter(x[i], y[i])\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:50:37.492215Z","iopub.execute_input":"2023-07-11T01:50:37.492922Z","iopub.status.idle":"2023-07-11T01:51:01.583260Z","shell.execute_reply.started":"2023-07-11T01:50:37.492893Z","shell.execute_reply":"2023-07-11T01:51:01.581992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 買い上げ点数が多いノードをベクトル可視化","metadata":{}},{"cell_type":"code","source":"buy_count = filtered_df.groupby(['t_dat', 'customer_id'])['article_id'].count().reset_index(name='item_buy_count')\n\ndf_buy_count = filtered_df.merge(buy_count, on=['t_dat', 'customer_id'], how='left')\ndf_buy_count = pd.merge(df_buy_count, articles[['article_id', 'product_type_name', 'product_group_name', 'graphical_appearance_name', 'colour_group_name', 'department_name', 'index_group_name']], on='article_id', how='left')\ndf_buy_count = pd.merge(df_buy_count, customers[['customer_id', 'age', 'club_member_status']], on='customer_id', how='left')\n\ndf_buy_count","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:51:01.585164Z","iopub.execute_input":"2023-07-11T01:51:01.585492Z","iopub.status.idle":"2023-07-11T01:51:15.732705Z","shell.execute_reply.started":"2023-07-11T01:51:01.585465Z","shell.execute_reply":"2023-07-11T01:51:15.731524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_buy_count['item_buy_count'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:51:15.737862Z","iopub.execute_input":"2023-07-11T01:51:15.738211Z","iopub.status.idle":"2023-07-11T01:51:15.858928Z","shell.execute_reply.started":"2023-07-11T01:51:15.738188Z","shell.execute_reply":"2023-07-11T01:51:15.857575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_buy_count2 = df_buy_count.copy()\ndf_buy_count2['t_dat'] = pd.to_datetime(df_buy_count2['t_dat'])\ndf_buy_count2 = df_buy_count.query('item_buy_count >= 50 and t_dat >= \"2019-04-01\"')\n\ndf_buy_count2","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:52:54.434802Z","iopub.execute_input":"2023-07-11T01:52:54.435217Z","iopub.status.idle":"2023-07-11T01:52:59.688930Z","shell.execute_reply.started":"2023-07-11T01:52:54.435191Z","shell.execute_reply":"2023-07-11T01:52:59.687861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"G = nx.Graph()\n\nfor index, row in tqdm(df_buy_count2.iterrows(), total=len(df_buy_count2)):\n    G.add_node(row['customer_id'], type='user')\n    G.add_node(row['article_id'], type='item')\n    G.add_edge(row['customer_id'], row['article_id'], weight=row['item_buy_count'])","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:52:59.690326Z","iopub.execute_input":"2023-07-11T01:52:59.690596Z","iopub.status.idle":"2023-07-11T01:53:01.137085Z","shell.execute_reply.started":"2023-07-11T01:52:59.690574Z","shell.execute_reply":"2023-07-11T01:53:01.135545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# ランダムウォークで取得したノードをWord2Vecに入力\n\nfrom gensim.models import Word2Vec as word2vec\n\nwalking = make_random_walks(G, 1024, 1024)\nmodel = word2vec(walking, min_count = 1, vector_size = 2, window = 10, workers = 4)\n\nx = list()\ny = list()\n# node_list = list()\nnode_dict = defaultdict(list)\ncolors = list()\nfig, ax = plt.subplots()\nfor node in tqdm(G.nodes):\n    #int型のままではイテレートできないので、string型に変換する\n    vector = model.wv[str(node)]  \n    x.append(vector[0])\n    y.append(vector[1])\n    node_dict[node] = [vector[0], vector[1]]\n    #注釈として、ノードの番号を追記する\n    #座標(x,y)は(vector[0],vector[1])を指定\n#     ax.annotate(str(node), (vector[0], vector[1]))\n#     if G.nodes[node][\"club\"] == \"Officer\":\n#         colors.append(\"r\")\n#     else:\n#         colors.append(\"b\")\nfor i in range(len(x)):\n#     ax.scatter(x[i], y[i], c=colors[i])\n    ax.scatter(x[i], y[i])\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:53:01.138442Z","iopub.execute_input":"2023-07-11T01:53:01.139231Z","iopub.status.idle":"2023-07-11T01:56:55.685425Z","shell.execute_reply.started":"2023-07-11T01:53:01.139204Z","shell.execute_reply":"2023-07-11T01:56:55.684434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"node_dict['c33fbb3dd8d249527cfc06231c1cc7a0703bab72d8e4030cb688ce7e5da0224f'][0]","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:56:55.687615Z","iopub.execute_input":"2023-07-11T01:56:55.687890Z","iopub.status.idle":"2023-07-11T01:56:55.695315Z","shell.execute_reply.started":"2023-07-11T01:56:55.687868Z","shell.execute_reply":"2023-07-11T01:56:55.693832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_buy_count2['customer_vec_x'] = df_buy_count2['customer_id'].progress_apply(lambda x: node_dict[x][0])\ndf_buy_count2['customer_vec_y'] = df_buy_count2['customer_id'].progress_apply(lambda x: node_dict[x][1])\ndf_buy_count2['article_vec_x'] = df_buy_count2['article_id'].progress_apply(lambda x: node_dict[x][0])\ndf_buy_count2['article_vec_y'] = df_buy_count2['article_id'].progress_apply(lambda x: node_dict[x][1])\n\ndf_buy_count2","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:56:55.696634Z","iopub.execute_input":"2023-07-11T01:56:55.696982Z","iopub.status.idle":"2023-07-11T01:56:55.882607Z","shell.execute_reply.started":"2023-07-11T01:56:55.696958Z","shell.execute_reply":"2023-07-11T01:56:55.880746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\nfig = px.scatter(df_buy_count2, x='article_vec_x' , y='article_vec_y', color=\"product_type_name\",\n                 size='item_buy_count', hover_data=['t_dat', 'age', 'club_member_status', 'product_type_name', 'product_group_name', 'graphical_appearance_name', 'colour_group_name', 'department_name', 'index_group_name'])\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:56:55.885111Z","iopub.execute_input":"2023-07-11T01:56:55.885492Z","iopub.status.idle":"2023-07-11T01:56:56.977124Z","shell.execute_reply.started":"2023-07-11T01:56:55.885467Z","shell.execute_reply":"2023-07-11T01:56:56.975999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig.write_html(\"/kaggle/working/H&M_item_vector_and_item_buy_count.html\")","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:56:56.978501Z","iopub.execute_input":"2023-07-11T01:56:56.978833Z","iopub.status.idle":"2023-07-11T01:56:57.356376Z","shell.execute_reply.started":"2023-07-11T01:56:56.978805Z","shell.execute_reply":"2023-07-11T01:56:57.354824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 18-22歳の顧客と買った商品についてベクトル可視化","metadata":{}},{"cell_type":"code","source":"df_buy_count3 = df_buy_count.copy()\ndf_buy_count3['t_dat'] = pd.to_datetime(df_buy_count3['t_dat'])\ndf_buy_count3 = df_buy_count.query('18 <= age <= 22 and item_buy_count <= 10 and t_dat >= \"2019-04-01\"')\n\ndf_buy_count3","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:56:57.358040Z","iopub.execute_input":"2023-07-11T01:56:57.358352Z","iopub.status.idle":"2023-07-11T01:56:59.061372Z","shell.execute_reply.started":"2023-07-11T01:56:57.358329Z","shell.execute_reply":"2023-07-11T01:56:59.059473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"G = nx.Graph()\n\nfor index, row in tqdm(df_buy_count3.iterrows(), total=len(df_buy_count3)):\n    G.add_node(row['customer_id'], type='user')\n    G.add_node(row['article_id'], type='item')\n    G.add_edge(row['customer_id'], row['article_id'], weight=row['item_buy_count'])","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:56:59.064791Z","iopub.execute_input":"2023-07-11T01:56:59.065160Z","iopub.status.idle":"2023-07-11T01:57:22.492030Z","shell.execute_reply.started":"2023-07-11T01:56:59.065136Z","shell.execute_reply":"2023-07-11T01:57:22.490387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# ランダムウォークで取得したノードをWord2Vecに入力\n\nfrom gensim.models import Word2Vec as word2vec\n\nwalking = make_random_walks(G, 1024, 1024)\nmodel = word2vec(walking, min_count = 1, vector_size = 2, window = 10, workers = 4)\n\nx = list()\ny = list()\n# node_list = list()\nnode_dict2 = defaultdict(list)\ncolors = list()\nfig, ax = plt.subplots()\nfor node in tqdm(G.nodes):\n    #int型のままではイテレートできないので、string型に変換する\n    vector = model.wv[str(node)]  \n    x.append(vector[0])\n    y.append(vector[1])\n    node_dict2[node] = [vector[0], vector[1]]\n    #注釈として、ノードの番号を追記する\n    #座標(x,y)は(vector[0],vector[1])を指定\n#     ax.annotate(str(node), (vector[0], vector[1]))\n#     if G.nodes[node][\"club\"] == \"Officer\":\n#         colors.append(\"r\")\n#     else:\n#         colors.append(\"b\")\nfor i in range(len(x)):\n#     ax.scatter(x[i], y[i], c=colors[i])\n    ax.scatter(x[i], y[i])\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-11T01:57:22.496583Z","iopub.execute_input":"2023-07-11T01:57:22.497017Z","iopub.status.idle":"2023-07-11T02:58:20.602558Z","shell.execute_reply.started":"2023-07-11T01:57:22.496990Z","shell.execute_reply":"2023-07-11T02:58:20.600978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_buy_count3['customer_vec_x'] = df_buy_count3['customer_id'].progress_apply(lambda x: node_dict2[x][0])\ndf_buy_count3['customer_vec_y'] = df_buy_count3['customer_id'].progress_apply(lambda x: node_dict2[x][1])\ndf_buy_count3['article_vec_x'] = df_buy_count3['article_id'].progress_apply(lambda x: node_dict2[x][0])\ndf_buy_count3['article_vec_y'] = df_buy_count3['article_id'].progress_apply(lambda x: node_dict2[x][1])\n\ndf_buy_count3","metadata":{"execution":{"iopub.status.busy":"2023-07-11T02:58:20.604590Z","iopub.execute_input":"2023-07-11T02:58:20.604907Z","iopub.status.idle":"2023-07-11T02:58:22.722486Z","shell.execute_reply.started":"2023-07-11T02:58:20.604883Z","shell.execute_reply":"2023-07-11T02:58:22.721182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\nfig = px.scatter(df_buy_count3, x='article_vec_x' , y='article_vec_y', color=\"product_type_name\",\n                 size='item_buy_count', hover_data=['t_dat', 'age', 'price', 'club_member_status', 'product_type_name', 'product_group_name', 'graphical_appearance_name', 'colour_group_name', 'department_name', 'index_group_name'])\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-11T03:02:55.964936Z","iopub.execute_input":"2023-07-11T03:02:55.965369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig.write_html(\"/kaggle/working/H&M_item_age_18_22_vector_and_item_buy_count.html\")","metadata":{"execution":{"iopub.status.busy":"2023-07-11T02:58:38.894754Z","iopub.execute_input":"2023-07-11T02:58:38.895102Z","iopub.status.idle":"2023-07-11T02:58:43.984476Z","shell.execute_reply.started":"2023-07-11T02:58:38.895078Z","shell.execute_reply":"2023-07-11T02:58:43.983301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}