{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n%matplotlib inline\n!pip install japanize-matplotlib\nimport japanize_matplotlib\nimport seaborn as sns\nfrom tqdm import tqdm # 進捗状況可視化用ライブラリ\ntqdm.pandas()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-19T23:54:06.107400Z","iopub.execute_input":"2023-07-19T23:54:06.108528Z","iopub.status.idle":"2023-07-19T23:54:21.588666Z","shell.execute_reply.started":"2023-07-19T23:54:06.108484Z","shell.execute_reply":"2023-07-19T23:54:21.587609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ncustomers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\ntransactions = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-19T23:54:21.591002Z","iopub.execute_input":"2023-07-19T23:54:21.591468Z","iopub.status.idle":"2023-07-19T23:55:38.894137Z","shell.execute_reply.started":"2023-07-19T23:54:21.591428Z","shell.execute_reply":"2023-07-19T23:55:38.893187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions['year'] = transactions['t_dat'].str[:4]\ntransactions['month'] = transactions['t_dat'].str[5:7]\ntransactions['year_month'] = transactions['t_dat'].str[:7]\n\ntransactions","metadata":{"execution":{"iopub.status.busy":"2023-07-19T23:55:38.913120Z","iopub.execute_input":"2023-07-19T23:55:38.913639Z","iopub.status.idle":"2023-07-19T23:56:11.570011Z","shell.execute_reply.started":"2023-07-19T23:55:38.913591Z","shell.execute_reply":"2023-07-19T23:56:11.568817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers['age2'] = customers['age'].apply(lambda x: x//10)\ncustomers","metadata":{"execution":{"iopub.status.busy":"2023-07-19T23:56:11.572991Z","iopub.execute_input":"2023-07-19T23:56:11.573478Z","iopub.status.idle":"2023-07-19T23:56:12.064602Z","shell.execute_reply.started":"2023-07-19T23:56:11.573438Z","shell.execute_reply":"2023-07-19T23:56:12.063693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# レコメンドモデル","metadata":{}},{"cell_type":"code","source":"%%time\n\n# まず、全体のトランザクションをt_dat × customer_idで集約したデータフレームを作成\n\ntransactions_sessions = transactions.groupby(['t_dat', 'customer_id'])['article_id'].apply(list).to_frame().reset_index()\ntransactions_sessions['month'] = transactions_sessions['t_dat'].str[5:7]\ntransactions_sessions","metadata":{"execution":{"iopub.status.busy":"2023-07-19T23:56:12.065681Z","iopub.execute_input":"2023-07-19T23:56:12.066181Z","iopub.status.idle":"2023-07-20T00:00:49.297717Z","shell.execute_reply.started":"2023-07-19T23:56:12.066154Z","shell.execute_reply":"2023-07-20T00:00:49.296654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 顧客リストの読み込み\n\ncustomer_list = pd.read_csv('/kaggle/input/h-m-customer-list-202004-202009/customer_list_202004_202009.csv').drop(columns='Unnamed: 0')\ncustomer_list['start_month'] = '04'\ncustomer_list['end_month'] = '09'\ncustomer_list","metadata":{"execution":{"iopub.status.busy":"2023-07-20T00:00:49.299279Z","iopub.execute_input":"2023-07-20T00:00:49.299944Z","iopub.status.idle":"2023-07-20T00:00:49.397122Z","shell.execute_reply.started":"2023-07-20T00:00:49.299913Z","shell.execute_reply":"2023-07-20T00:00:49.395977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 各セッションで、対象の商品が購入されているかチェックするフラグを作成する関数\n\ndef check_transactions(items: list, target_items: list):\n    result = 0\n    for item in items:\n        if item in target_items:\n            result = 1\n    return result","metadata":{"execution":{"iopub.status.busy":"2023-07-20T00:00:49.398639Z","iopub.execute_input":"2023-07-20T00:00:49.399054Z","iopub.status.idle":"2023-07-20T00:00:49.404606Z","shell.execute_reply.started":"2023-07-20T00:00:49.399014Z","shell.execute_reply":"2023-07-20T00:00:49.403791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 画像の類似度を測るためのライブラリ\n\n!pip install imgsim\nimport imgsim\nvtr = imgsim.Vectorizer()","metadata":{"execution":{"iopub.status.busy":"2023-07-20T00:00:49.405590Z","iopub.execute_input":"2023-07-20T00:00:49.406441Z","iopub.status.idle":"2023-07-20T00:01:29.515097Z","shell.execute_reply.started":"2023-07-20T00:00:49.406411Z","shell.execute_reply":"2023-07-20T00:01:29.513700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mlxtend.preprocessing import TransactionEncoder\nfrom mlxtend.frequent_patterns import apriori\nfrom mlxtend.frequent_patterns import association_rules\n\nimport glob\nimport cv2\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nfrom collections import defaultdict, Counter\n\n\nclass Recommend_By_Combination:\n    \n    def __init__(self, target_transactions: pd.DataFrame, row_index: int):\n        \"\"\"\n        引数\n        target_transactions: 顧客リストのデータフレーム('customer_id', 'article_id', 'month', t_dat×customer_idごとの'article_id'のリストが必要)\n        row_index: 顧客リストの中の行\n        \"\"\"\n        \n        self.target_transactions = target_transactions\n        self.row_index = row_index\n        self.customer_id = self.target_transactions.loc[self.target_transactions.index == self.row_index]['customer_id'].values[0]\n#         self.article_id = self.target_transactions.loc[self.target_transactions.index == self.row_index]['article_id'].values[0]\n        \n#         self.month = self.target_transactions.loc[self.target_transactions.index == self.row_index]['month'].astype('int').values[0]\n#         self.season_start = self.month // 4 * 3 + 1 # 対象にしたい季節の初めの月\n#         self.season_end = self.month // 4 * 3 + 3 # 対象にしたい季節の終わりの月\n        self.season_start = self.target_transactions.loc[self.target_transactions.index == self.row_index]['start_month'].astype('int').values[0]\n        self.season_end = self.target_transactions.loc[self.target_transactions.index == self.row_index]['end_month'].astype('int').values[0]\n        # 対象期間\n        if self.season_end < self.season_start: # 期間の途中に12月, 1月を含む場合\n            self.target_season = np.concatenate(np.arange(self.season_start, 12+1), np.arange(1, self.season_end+1)) # 12月までの月と、1月以降の月の配列\n        else:\n            self.target_season = np.arange(self.season_start, self.season_end+1)\n        print(f'対象期間: {self.target_season}')\n        \n        # 対照の顧客が、期間内に一番直近で購入した商品\n#         self.article_id = transactions.loc[(transactions['customer_id'] == self.customer_id) & (self.season_start <= transactions['month'].astype('int')) & (transactions['month'].astype('int') <= self.season_end)].tail(1)['article_id'].values[0]\n        self.article_id = transactions.loc[(transactions['customer_id'] == self.customer_id) & (transactions['month'].astype('int').apply(lambda x: x in self.target_season))].tail(1)['article_id'].values[0]\n        \n        # 過去に購入した商品一覧(顧客idと、季節が一致する行)\n#         self.past_items = transactions.loc[(transactions['customer_id'] == self.customer_id) & (self.season_start <= transactions['month'].astype('int')) & (transactions['month'].astype('int') <= self.season_end)]['article_id'].to_list() \n        self.past_items = transactions.loc[(transactions['customer_id'] == self.customer_id) & (transactions['month'].astype('int').apply(lambda x: x in self.target_season))]['article_id'].to_list() \n    \n        # self.article_idと同じ商品を購入していて、かつ季節が同じセッション(同じ商品を購入した、他の顧客も含んだもの)\n        transactions_sessions['target_item_flg'] = transactions_sessions['article_id'].progress_apply(lambda x: self.article_id in x)\n#         self.target_sessions = transactions_sessions.query('target_item_flg == True').loc[(self.season_start <= transactions_sessions['month'].astype('int')) & (transactions_sessions['month'].astype('int') <= self.season_end)]\n        self.target_sessions = transactions_sessions.query('target_item_flg == True').loc[transactions_sessions['month'].astype('int').apply(lambda x: x in self.target_season)]\n        \n        # 上記のセッションで買われた商品一覧\n        self.target_sessions_items_list = self.target_sessions['article_id'].to_list()\n        \n        # 該当の顧客が過去に買った商品と、同じものを購入していて、季節が同じセッション\n        transactions_sessions['target_item_flg2'] = transactions_sessions['article_id'].progress_apply(lambda x: check_transactions(x, self.past_items))\n#         self.target_sessions2 = transactions_sessions.query('target_item_flg2 == True').loc[(self.season_start <= transactions_sessions['month'].astype('int')) & (transactions_sessions['month'].astype('int') <= self.season_end)]\n        self.target_sessions2 = transactions_sessions.query('target_item_flg2 == True').loc[transactions_sessions['month'].astype('int').apply(lambda x: x in self.target_season)]\n        \n        # 上記のセッションで買われた商品一覧\n        self.target_sessions_past_items_list = self.target_sessions2['article_id'].to_list()\n        \n        \n        print(f'顧客: {self.customer_id}, 購入商品: {self.article_id}')\n        print()\n        print('======== 同じ商品を購入しているセッション ========')\n        display(self.target_sessions.head())\n        \n        print()\n        print('======== 対象となる商品 ========')\n        self.target_file = glob.glob(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/*/0{self.article_id}.jpg\")\n        self.target_img = cv2.imread(self.target_file[0])\n        self.target_img = cv2.cvtColor(self.target_img, cv2.COLOR_BGR2RGB) \n        plt.imshow(self.target_img) \n        plt.show()\n        \n        print()\n        print('======== 過去の同時期に購入した商品 ========')\n        self.past_files = []\n        for id in self.past_items:\n            self.past_files += glob.glob(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/*/0{id}.jpg\")\n        \n        fig = plt.figure(1, (30., 30.))\n        grid = ImageGrid(fig, 111,\n                 nrows_ncols=(len(self.past_files) // 10 + 1, 10),\n                 axes_pad=0.1)  \n\n        for n in tqdm(range(len(self.past_files))):\n            img = cv2.imread(self.past_files[n])\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB) \n            grid[n].imshow(img) \n        \n    def association_analysis(self, min_support=0.01, max_len=3, use_past=False):\n        \"\"\"\n        アソシエーション分析で、与えられた引数から有効な商品の組み合わせを表示する関数\n        \n        引数 \n        min_support: 支持度の最小値, max_len: 組み合わせの上限個数\n        max_len: 生成したいitemsetsの個数上限\n        use_past: レコメンドする際に、過去の購入歴も考慮するか\n        \"\"\"\n        \n        # データをテーブル形式に加工\n        self.te = TransactionEncoder()\n        \n        if use_past==False:\n            self.te_array = self.te.fit(self.target_sessions_items_list).transform(self.target_sessions_items_list)\n        else: # 過去の購入歴からレコメンドしたい場合\n            self.te_array = self.te.fit(self.target_sessions_past_items_list).transform(self.target_sessions_past_items_list)\n            \n        self.df_tmp   = pd.DataFrame(self.te_array, columns=self.te.columns_)\n\n        # アソシエーション分析で、一緒に買われやすい組み合わせを調べる\n        self.freq_items = apriori(self.df_tmp,                   # データフレーム\n                             min_support  = min_support,    # 支持度(support)の最小値\n                             use_colnames = True,           # 出力値のカラムに購入商品名を表示\n                             max_len      = max_len,        # 生成されるitemsetsの個数上限\n                             verbose = 0,                   # low_memory=Trueの場合のイテレーション数\n                             low_memory = False,            # メモリ制限あり＆大規模なデータセット利用時に有効\n                            )\n        \n        # 支持度が高い順に結果出力\n        self.freq_items = self.freq_items.sort_values(\"support\", ascending = False).reset_index(drop=True)\n        print(self.freq_items)\n        \n    def association_rules(self, metric=\"confidence\", min_threshold=0.1, set_num=0):\n        \"\"\"\n        アソシエーション分析の結果から、リフト値や各商品ごとの支持度など、詳細な情報を取得する関数\n        \n        引数\n        metric: アソシエーション・ルールの評価指標。support(支持度), confidence, lift(リフト値)など\n        min_threshold: metricsの閾値\n        set_num: 何個の組み合わせでレコメンドするか。入力がない場合は、抽出したもののうち最も多い個数\n        \"\"\"\n\n        # アソシエーション・ルール抽出\n        self.df_rules = association_rules(self.freq_items,             # supportとitemsetsを持つデータフレーム\n                                     metric = \"confidence\",  # アソシエーション・ルールの評価指標\n                                     min_threshold = 0.1,    # metricsの閾値\n                                    )\n        \n        # リフト値、支持度の順で抽出結果を並び替え\n        # 参考: https://www.zetta.co.jp/bigdata/l_07.shtml\n        # antecedents: 購入した商品 consequents: 次に買われやすい商品\n        self.df_rules_sorted = self.df_rules.sort_values(['lift', 'support'], ascending=False)\n        \n        print()\n        print('======== アソシエーションルール抽出後のデータフレーム(リフト値、支持度の高い順) ========')\n        display(self.df_rules_sorted)\n        \n        # 商品の組み合わせのリストを作成\n        self.df_rules_sorted['items_list'] = self.df_rules_sorted[['antecedents', 'consequents']].progress_apply(lambda x: list(x[0]) + list(x[1]), axis=1)\n        \n        # 後に表示しやすいよう、商品の組み合わせの個数を決定する\n        self.df_rules_sorted['set_items_num'] = self.df_rules_sorted['items_list'].map(len)\n        if set_num == 0: # 引数に対する入力がなかった場合は、可能な最大数\n            self.set_num = self.df_rules_sorted['set_items_num'].max()\n        else:\n            self.set_num = set_num\n        self.df_sets = self.df_rules_sorted.loc[self.df_rules_sorted['set_items_num'] == self.set_num]\n        \n        print()\n        print('======== 表示する対象の商品群 ========')\n        display(self.df_sets)\n        \n    def recommend_items(self, show_max_length=20):\n        \"\"\"\n        購入した商品と、一緒に買われやすい商品の組み合わせの画像を表示して、レコメンドする関数\n        \n        引数\n        show_max_length: 表示したい最大商品数\n        \"\"\"\n\n        self.TARGET_ITEMS = self.df_sets['items_list'].head(show_max_length).to_list()\n        print(f'レコメンドする商品: {self.TARGET_ITEMS}')\n\n        # ファイルの読み込み\n        self.files = []\n        self.item_names = []\n        for combination in tqdm(self.TARGET_ITEMS):\n            for id in combination:\n                self.files += glob.glob(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/*/0{id}.jpg\")\n                self.item_names.append(articles.query('article_id == @id')['prod_name'].values[0])\n\n        # 読み込んだファイルの可視化\n        %matplotlib inline\n        fig = plt.figure(1, (60., 60.))\n        grid = ImageGrid(fig, 111,\n                         nrows_ncols=(len(self.files) // self.set_num + 1, self.set_num),\n                         axes_pad=0.1)    \n\n        self.imgs = []\n        for n in tqdm(range(len(self.files))):\n            img = cv2.imread(self.files[n])\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB) \n            grid[n].imshow(img) \n\n            # 配列に画像情報を格納\n            img = cv2.resize(img, (256, 256))\n            img_expanded = np.expand_dims(img, axis=0)\n            self.imgs.append(img_expanded)\n        #     print(imgs)\n\n        self.imgs = np.concatenate(self.imgs,axis=0)\n        \n        return self.TARGET_ITEMS\n        \n    def other_recommends(self, type_kind=\"product_type_name\", show_max_length=20):\n        \"\"\"\n        同じ種類の商品の中から、似ている商品をレコメンドする関数\n        \n        引数\n        type_kind: どのジャンルが同じ商品をレコメンドしたいか\n        show_max_length: 表示したい最大商品数\n        \"\"\"\n        \n        self.item_type = articles.loc[articles['article_id'] == self.article_id][type_kind].values[0]\n        self.same_type_items = articles.loc[articles[type_kind] == self.item_type]['article_id'].to_list()\n        \n        # ファイルの読み込み\n        self.same_type_files = defaultdict(list)\n        self.same_type_item_names = defaultdict(list)\n        for id in tqdm(self.same_type_items):\n            self.same_type_files[id] += glob.glob(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/*/0{id}.jpg\")\n            self.same_type_item_names[id].append(articles.query('article_id == @id')['prod_name'].values[0])\n            \n        self.same_type_imgs = defaultdict(list)\n        self.same_type_item_vecs = defaultdict(list)\n\n        # 辞書に同じジャンルの商品の各画像やベクトルを格納\n        for id, file in tqdm(self.same_type_files.items()):\n            try:\n                img = cv2.imread(file[0])\n                img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB) \n#                 img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) \n            # 明るさ・コントラスト操作\n        #     img = cv2.convertScaleAbs(img,alpha = 1.3,beta = 60)\n                img = cv2.resize(img, (256, 256))\n#                 img_expanded = np.expand_dims(img, axis=0)\n                self.same_type_imgs[id].append(img)\n                self.same_type_item_vecs[id] = vtr.vectorize(img)\n            except:\n                pass\n\n        self.similarities_dict = defaultdict(list)\n        for id in tqdm(self.same_type_items):\n            try:\n                self.similarities_dict[id] = imgsim.distance(self.same_type_item_vecs[self.article_id], self.same_type_item_vecs[id])\n            except:\n                pass\n            \n        self.df_similarities = pd.DataFrame.from_dict(self.similarities_dict, orient=\"index\", columns=[self.article_id])\n        display(self.df_similarities.sort_values(self.article_id))\n        \n        # レコメンドする商品を表示\n        %matplotlib inline\n        fig = plt.figure(1, (60., 60.))\n        grid = ImageGrid(fig, 111,\n                         nrows_ncols=(show_max_length // 5 + 1, 5),\n                         axes_pad=0.1) \n        \n        n = 0\n        self.TARGET_ITEMS2 = self.df_similarities.sort_values(self.article_id).head(show_max_length).index.to_list()\n        for id in self.TARGET_ITEMS2:\n            grid[n].imshow(self.same_type_imgs[id][0])\n            n += 1\n            \n        print(f'レコメンドする商品: {self.TARGET_ITEMS2}')\n        return self.TARGET_ITEMS2\n            \n    def recommend_by_frequent_words(self, stop_words=['a', 'and', 'with', 'the', 'at', 'in'], type_kind=\"product_type_name\", frequent_words_num=5):\n        \"\"\"\n    　　対象顧客が、過去に購入した商品の中で、頻発する言葉と関連する人気商品をレコメンド\n        \"\"\"\n        \n        # 直近に買った商品のジャンル\n        self.item_type2 = articles.loc[articles['article_id'] == self.article_id][type_kind].values[0]\n        \n        # 対象顧客が買った商品の説明に含まれる単語リスト\n        self.customer_words = ','.join(articles.loc[articles['article_id'].apply(lambda x: x in self.past_items)]['detail_desc'].values.flatten()).split(' ')\n        # ストップワードを除去\n        self.customer_words = [word for word in self.customer_words if word not in stop_words]\n#         print(self.customer_words)\n\n        # 各単語の頻度を集計\n        self.words_frequency = Counter(self.customer_words)\n        print(self.words_frequency)\n        \n        # 頻出語\n        self.frequent_word = self.words_frequency.most_common()[:frequent_words_num][:][0]\n        print(self.frequent_word)\n        \n        # 直近で購入した商品と同じジャンルの商品、又は同じ商品を買ったセッションで買われている商品\n        self.target_tmp = articles.loc[(articles['article_id'].apply(lambda x: x in self.target_sessions_past_items_list)) | (articles[type_kind] == self.item_type2)]\n        \n        # そのうち、頻出語を説明文に含む商品\n        self.target_items_df = self.target_tmp.loc[self.target_tmp['detail_desc'].str.contains('|'.join(self.frequent_word))]\n        self.TARGET_ITEMS3 = self.target_items_df['article_id'].to_list()\n        self.target_items_details = self.target_items_df['detail_desc'].to_list()\n        \n        print(self.target_items_details)\n        \n        \n        \n    def validate(self, recommended_items: list):\n        \"\"\"\n        レコメンド結果が、どのくらい予測精度があるか評価する関数\n        \n        引数\n        recommended_items: レコメンドした商品のidリスト\n        \"\"\"\n        \n        self.recommended_items = np.array(recommended_items).flatten()\n        \n        # 実際に対象顧客が買った商品(今回は、対象期間外の時期に、過去に買っている商品とする)\n        self.true_items = transactions.loc[(transactions['customer_id'] == self.customer_id) & (transactions_sessions['month'].astype('int').apply(lambda x: x not in self.target_season))]['article_id'].to_list() \n        \n        # 真陽性(レコメンドしたもので、実際に購入していた商品)の数\n        self.true_positive = len(set(self.true_items) & set(self.recommended_items))\n        \n        # 偽陽性の数(レコメンドしたが、実際には購入していなかった商品)の数\n        self.false_positive = len(set(self.recommended_items)) - self.true_positive\n        \n        # 偽陰性の数(レコメンドしなかったが、実際には購入していた商品)の数\n        self.false_negative = len(set(self.true_items)) - self.true_positive\n        \n#         print(self.true_items, self.true_positive, self.false_positive, self.false_negative)\n        \n        # 再現率(実際に購入した商品の中で、レコメンドできた割合)\n        # TP / (TP + FN)\n        self.recall =  self.true_positive / (self.true_positive + self.false_negative)\n        \n        # 適合率(レコメンドした中で、実際に購入していた割合)\n        # TP / (TP + FP)\n        self.precision = self.true_positive / (self.true_positive + self.false_positive)\n        \n        # F値(適合率と再現率の調和平均。高いほど良い)\n        # (2*適合率*再現率) / (適合率+再現率)\n        self.F_score = 2 * self.precision * self.recall / (self.precision + self.recall + 1e-10)\n        \n        # 結果をまとめたデータフレームを作成\n        \n        ## 真陽性、偽陽性、偽陰性を表にまとめる\n        self.result_matrix = pd.DataFrame(columns=['購入した', '購入しなかった'], index=['レコメンドした', 'レコメンドしなかった'])\n        self.result_matrix.loc['レコメンドした', '購入した'] = self.true_positive\n        self.result_matrix.loc['レコメンドした', '購入しなかった'] = self.false_positive\n        self.result_matrix.loc['レコメンドしなかった', '購入した'] = self.false_negative\n        \n        ## 再現率、適合率、F値をまとめる\n        self.result_df = pd.DataFrame(\n            data = {'再現率': [self.recall],\n                    '適合率': [self.precision],\n                    'F値': [self.F_score]\n                    }\n        )\n        \n        print('=========================')\n        print('真陽性、偽陽性、偽陰性')\n        print('========================')\n        display(self.result_matrix)\n        print('')\n        print('=========================')\n        print('再現率、適合率、F値')\n        print('========================')\n        display(self.result_df)\n        \n        \n        ","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:50:43.515750Z","iopub.execute_input":"2023-07-20T02:50:43.516227Z","iopub.status.idle":"2023-07-20T02:50:43.627280Z","shell.execute_reply.started":"2023-07-20T02:50:43.516194Z","shell.execute_reply":"2023-07-20T02:50:43.626158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = Recommend_By_Combination(transactions, 124)\nmodel = Recommend_By_Combination(customer_list, 0)","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:50:44.562047Z","iopub.execute_input":"2023-07-20T02:50:44.562497Z","iopub.status.idle":"2023-07-20T02:57:05.680993Z","shell.execute_reply.started":"2023-07-20T02:50:44.562460Z","shell.execute_reply":"2023-07-20T02:57:05.679676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.association_analysis(min_support=0.01, max_len=3,use_past=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-20T00:07:56.065715Z","iopub.execute_input":"2023-07-20T00:07:56.066604Z","iopub.status.idle":"2023-07-20T00:07:56.880925Z","shell.execute_reply.started":"2023-07-20T00:07:56.066565Z","shell.execute_reply":"2023-07-20T00:07:56.879829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.association_rules()","metadata":{"execution":{"iopub.status.busy":"2023-07-20T00:07:56.882404Z","iopub.execute_input":"2023-07-20T00:07:56.883135Z","iopub.status.idle":"2023-07-20T00:07:56.937477Z","shell.execute_reply.started":"2023-07-20T00:07:56.883104Z","shell.execute_reply":"2023-07-20T00:07:56.936235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommend_items1 = model.recommend_items(show_max_length=50)\nrecommend_items1","metadata":{"execution":{"iopub.status.busy":"2023-07-20T00:07:56.938942Z","iopub.execute_input":"2023-07-20T00:07:56.940304Z","iopub.status.idle":"2023-07-20T00:08:06.883921Z","shell.execute_reply.started":"2023-07-20T00:07:56.940269Z","shell.execute_reply":"2023-07-20T00:08:06.881681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommend_items2 = model.other_recommends(type_kind='index_name', show_max_length=20)\nrecommend_items2","metadata":{"execution":{"iopub.status.busy":"2023-07-20T00:08:06.885291Z","iopub.execute_input":"2023-07-20T00:08:06.885813Z","iopub.status.idle":"2023-07-20T02:28:06.462938Z","shell.execute_reply.started":"2023-07-20T00:08:06.885784Z","shell.execute_reply":"2023-07-20T02:28:06.460018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.recommend_by_frequent_words()","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:57:52.755904Z","iopub.execute_input":"2023-07-20T02:57:52.756204Z","iopub.status.idle":"2023-07-20T02:59:09.438665Z","shell.execute_reply.started":"2023-07-20T02:57:52.756178Z","shell.execute_reply":"2023-07-20T02:59:09.437153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.validate(recommend_items2)","metadata":{"execution":{"iopub.status.busy":"2023-07-20T03:03:25.813503Z","iopub.execute_input":"2023-07-20T03:03:25.813984Z","iopub.status.idle":"2023-07-20T03:04:12.209202Z","shell.execute_reply.started":"2023-07-20T03:03:25.813952Z","shell.execute_reply":"2023-07-20T03:04:12.208444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(recommend_items2)","metadata":{"execution":{"iopub.status.busy":"2023-07-20T03:04:12.210686Z","iopub.execute_input":"2023-07-20T03:04:12.211382Z","iopub.status.idle":"2023-07-20T03:04:12.216442Z","shell.execute_reply.started":"2023-07-20T03:04:12.211342Z","shell.execute_reply":"2023-07-20T03:04:12.215403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(recommend_items1).flatten()","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.276432Z","iopub.status.idle":"2023-07-20T02:29:23.277177Z","shell.execute_reply.started":"2023-07-20T02:29:23.276954Z","shell.execute_reply":"2023-07-20T02:29:23.276980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'|'.join(['a', 'b', 'c'])","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:34:41.762594Z","iopub.execute_input":"2023-07-20T02:34:41.763015Z","iopub.status.idle":"2023-07-20T02:34:41.771573Z","shell.execute_reply.started":"2023-07-20T02:34:41.762984Z","shell.execute_reply":"2023-07-20T02:34:41.770310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.same_type_imgs","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.278697Z","iopub.status.idle":"2023-07-20T02:29:23.279108Z","shell.execute_reply.started":"2023-07-20T02:29:23.278901Z","shell.execute_reply":"2023-07-20T02:29:23.278920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.same_type_files","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.280600Z","iopub.status.idle":"2023-07-20T02:29:23.281020Z","shell.execute_reply.started":"2023-07-20T02:29:23.280802Z","shell.execute_reply":"2023-07-20T02:29:23.280821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for id, file in model.same_type_files.items():\n    print(id)\n    print(file[0])","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.283017Z","iopub.status.idle":"2023-07-20T02:29:23.283770Z","shell.execute_reply.started":"2023-07-20T02:29:23.283573Z","shell.execute_reply":"2023-07-20T02:29:23.283593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.same_type_items","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.285125Z","iopub.status.idle":"2023-07-20T02:29:23.285586Z","shell.execute_reply.started":"2023-07-20T02:29:23.285340Z","shell.execute_reply":"2023-07-20T02:29:23.285379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.files","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.286972Z","iopub.status.idle":"2023-07-20T02:29:23.287379Z","shell.execute_reply.started":"2023-07-20T02:29:23.287156Z","shell.execute_reply":"2023-07-20T02:29:23.287172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # target_file = glob.glob(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/*/0580482004.jpg\")\n# target_file = glob.glob(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/*/0663713001.jpg\")\n# target_img = cv2.imread(target_file[0])\n# target_img = cv2.cvtColor(target_img, cv2.COLOR_BGR2RGB) \n# plt.imshow(target_img) \n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.288875Z","iopub.status.idle":"2023-07-20T02:29:23.289261Z","shell.execute_reply.started":"2023-07-20T02:29:23.289074Z","shell.execute_reply":"2023-07-20T02:29:23.289092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_file","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.290833Z","iopub.status.idle":"2023-07-20T02:29:23.291393Z","shell.execute_reply.started":"2023-07-20T02:29:23.291102Z","shell.execute_reply":"2023-07-20T02:29:23.291128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_img","metadata":{"execution":{"iopub.status.busy":"2023-07-20T02:29:23.293002Z","iopub.status.idle":"2023-07-20T02:29:23.293416Z","shell.execute_reply.started":"2023-07-20T02:29:23.293187Z","shell.execute_reply":"2023-07-20T02:29:23.293204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}