{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n!pip install --upgrade implicit","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-06T09:18:46.32538Z","iopub.execute_input":"2022-04-06T09:18:46.32621Z","iopub.status.idle":"2022-04-06T09:18:58.730128Z","shell.execute_reply.started":"2022-04-06T09:18:46.326153Z","shell.execute_reply":"2022-04-06T09:18:58.728159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os; os.environ['OPENBLAS_NUM_THREADS']='1'\nimport numpy as np\nimport pandas as pd\nimport implicit\nfrom scipy.sparse import coo_matrix\nfrom implicit.evaluation import mean_average_precision_at_k","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:18:58.733347Z","iopub.execute_input":"2022-04-06T09:18:58.7338Z","iopub.status.idle":"2022-04-06T09:18:58.742179Z","shell.execute_reply.started":"2022-04-06T09:18:58.733755Z","shell.execute_reply":"2022-04-06T09:18:58.740878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nbase_path = '../input/h-and-m-personalized-fashion-recommendations/'\ncsv_train = f'{base_path}transactions_train.csv'\ncsv_sub = f'{base_path}sample_submission.csv'\ncsv_users = f'{base_path}customers.csv'\ncsv_items = f'{base_path}articles.csv'\n\ndf = pd.read_csv(csv_train, dtype={'article_id': str}, parse_dates=['t_dat'])\ndf_sub = pd.read_csv(csv_sub)\ndfu = pd.read_csv(csv_users)\ndfi = pd.read_csv(csv_items, dtype={'article_id': str})","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:18:58.743963Z","iopub.execute_input":"2022-04-06T09:18:58.744259Z","iopub.status.idle":"2022-04-06T09:20:18.9274Z","shell.execute_reply.started":"2022-04-06T09:18:58.744197Z","shell.execute_reply":"2022-04-06T09:20:18.924883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['t_dat']","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:18.932915Z","iopub.execute_input":"2022-04-06T09:20:18.935112Z","iopub.status.idle":"2022-04-06T09:20:18.966569Z","shell.execute_reply.started":"2022-04-06T09:20:18.93491Z","shell.execute_reply":"2022-04-06T09:20:18.964296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Trying with less data(少ないデータで試す):\n# https://www.kaggle.com/tomooinubushi/folk-of-time-is-our-best-friend/notebook\ndf = df[df['t_dat'] > '2020-08-21']\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:18.968613Z","iopub.execute_input":"2022-04-06T09:20:18.969121Z","iopub.status.idle":"2022-04-06T09:20:20.365907Z","shell.execute_reply.started":"2022-04-06T09:20:18.969036Z","shell.execute_reply":"2022-04-06T09:20:20.364724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For validation this means 3 weeks of training and 1 week for validation\n  #バリデーションの場合は、トレーニングに3週間、バリデーションに1週間ということになります。\n# For submission, it means 4 weeks of training\n  #投稿の場合は、4週間のトレーニングということになります\ndf['t_dat'].max()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:20.367314Z","iopub.execute_input":"2022-04-06T09:20:20.367529Z","iopub.status.idle":"2022-04-06T09:20:20.381641Z","shell.execute_reply.started":"2022-04-06T09:20:20.367505Z","shell.execute_reply":"2022-04-06T09:20:20.380112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign autoincrementing ids starting from 0 to both users and items\n# ユーザーとアイテムの両方に0から始まる自動インクリメントのidを割り当てる。\n\n#ンクリメントとは、増加、増分などの意味の英単語だが、コンピュータでは数値に1を加える操作のことを指す","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:20.383977Z","iopub.execute_input":"2022-04-06T09:20:20.384417Z","iopub.status.idle":"2022-04-06T09:20:20.391665Z","shell.execute_reply.started":"2022-04-06T09:20:20.384376Z","shell.execute_reply":"2022-04-06T09:20:20.390878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:20.392948Z","iopub.execute_input":"2022-04-06T09:20:20.393181Z","iopub.status.idle":"2022-04-06T09:20:20.418039Z","shell.execute_reply.started":"2022-04-06T09:20:20.393156Z","shell.execute_reply":"2022-04-06T09:20:20.417122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ALL_USERS = dfu['customer_id'].unique().tolist()\nALL_ITEMS = dfi['article_id'].unique().tolist()\n\n#上では、それぞれのカスタマー名と商品名の固有の値を抽出させ、リストに変換させている\n\nuser_ids = dict(list(enumerate(ALL_USERS)))\nitem_ids = dict(list(enumerate(ALL_ITEMS)))\n\n# enumerate()を用いることにより、インデックス番号, 要素と取得することが出来る。\n #商品、アイテム名を番号処理することが可能。\n\nuser_map = {u: uidx for uidx, u in user_ids.items()}\nitem_map = {i: iidx for iidx, i in item_ids.items()}\n\ndf['user_id'] = df['customer_id'].map(user_map)\ndf['item_id'] = df['article_id'].map(item_map)\n\ndel dfu, dfi","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:20.419529Z","iopub.execute_input":"2022-04-06T09:20:20.420384Z","iopub.status.idle":"2022-04-06T09:20:23.492881Z","shell.execute_reply.started":"2022-04-06T09:20:20.420308Z","shell.execute_reply":"2022-04-06T09:20:23.492026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()\n\n##これで、ユーザーIDとアイテムIDを扱い安やすくすることができた。","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:23.495253Z","iopub.execute_input":"2022-04-06T09:20:23.496273Z","iopub.status.idle":"2022-04-06T09:20:23.510167Z","shell.execute_reply.started":"2022-04-06T09:20:23.496169Z","shell.execute_reply":"2022-04-06T09:20:23.509308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create coo_matrix (user x item) and csr matrix (user x item)\n#coo_matrix (user x item) と csr matrix (user x item) を作成する。","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:23.512272Z","iopub.execute_input":"2022-04-06T09:20:23.512546Z","iopub.status.idle":"2022-04-06T09:20:23.528539Z","shell.execute_reply.started":"2022-04-06T09:20:23.51252Z","shell.execute_reply":"2022-04-06T09:20:23.526863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = df['user_id'].values\ncol = df['item_id'].values\n\nprint(row)\nprint(col)\ndata = np.ones(df.shape[0])\ncoo_train = coo_matrix((data, (row, col)), shape=(len(ALL_USERS), len(ALL_ITEMS)))\ncoo_train","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:23.530389Z","iopub.execute_input":"2022-04-06T09:20:23.530682Z","iopub.status.idle":"2022-04-06T09:20:23.566897Z","shell.execute_reply.started":"2022-04-06T09:20:23.530646Z","shell.execute_reply":"2022-04-06T09:20:23.566077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_cut = df['t_dat'].max() - pd.Timedelta(7,\"d\")\nprint(validation_cut)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:23.568024Z","iopub.execute_input":"2022-04-06T09:20:23.569046Z","iopub.status.idle":"2022-04-06T09:20:23.58246Z","shell.execute_reply.started":"2022-04-06T09:20:23.568989Z","shell.execute_reply":"2022-04-06T09:20:23.581166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Timedelta(7, \"d\")","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:23.583883Z","iopub.execute_input":"2022-04-06T09:20:23.584167Z","iopub.status.idle":"2022-04-06T09:20:23.596674Z","shell.execute_reply.started":"2022-04-06T09:20:23.584135Z","shell.execute_reply":"2022-04-06T09:20:23.594412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['t_dat'].max())\nprint(validation_cut)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:23.599254Z","iopub.execute_input":"2022-04-06T09:20:23.599582Z","iopub.status.idle":"2022-04-06T09:20:23.617488Z","shell.execute_reply.started":"2022-04-06T09:20:23.599551Z","shell.execute_reply":"2022-04-06T09:20:23.616284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel = implicit.als.AlternatingLeastSquares(factors=10, iterations=2)\nmodel.fit(coo_train)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:23.619866Z","iopub.execute_input":"2022-04-06T09:20:23.620199Z","iopub.status.idle":"2022-04-06T09:20:26.06884Z","shell.execute_reply.started":"2022-04-06T09:20:23.620161Z","shell.execute_reply":"2022-04-06T09:20:26.068346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:26.070115Z","iopub.execute_input":"2022-04-06T09:20:26.07047Z","iopub.status.idle":"2022-04-06T09:20:26.083239Z","shell.execute_reply.started":"2022-04-06T09:20:26.070442Z","shell.execute_reply":"2022-04-06T09:20:26.081804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row  = np.array([0,3,1,0]) # user_ids\ncol  = np.array([0,3,1,2]) # item_ids\ndata = np.array([4,5,7,9]) # a bunch of ones of lenght unique(user) x unique(items)\ncoo_matrix((data,(row,col)), shape=(4,4)).todense()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:26.084934Z","iopub.execute_input":"2022-04-06T09:20:26.085194Z","iopub.status.idle":"2022-04-06T09:20:26.108995Z","shell.execute_reply.started":"2022-04-06T09:20:26.085161Z","shell.execute_reply":"2022-04-06T09:20:26.108025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coo_matrix((data,(row,col)), shape=(4,4))","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:26.112242Z","iopub.execute_input":"2022-04-06T09:20:26.112544Z","iopub.status.idle":"2022-04-06T09:20:26.133812Z","shell.execute_reply.started":"2022-04-06T09:20:26.112516Z","shell.execute_reply":"2022-04-06T09:20:26.132626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_user_item_coo(df):\n    \"Turn a dataframe with transactions into a COO sparse items x users matrix\"\n    \" トランザクションを含むデータフレームをCOOスパースアイテム×ユーザー行列に変換する\"\n    row = df['user_id'].values\n    col = df['item_id'].values\n    data = np.ones(df.shape[0])\n    coo = coo_matrix((data, (row, col)), shape=(len(ALL_USERS), len(ALL_ITEMS)))\n    return coo\n\n\ndef split_data(df, validation_days=7):\n    \"\"\" Split a pandas dataframe into training and validation data, using <<validation_days>>\n    pandas のデータフレームを <<validation_days で学習データと検証データに分割する\n    \"\"\"\n    validation_cut = df['t_dat'].max() - pd.Timedelta(validation_days,\"d\")\n    df_train = df[df['t_dat'] < validation_cut]\n    df_val = df[df['t_dat'] >= validation_cut]\n    return df_train, df_val\n    \n\ndef get_val_matrices(df, validation_days=7):\n    \"\"\" Split into training and validation and create various matrices\n    トレーニング用と検証用に分割し、各種マトリクスを作成する\n        \n         Returns a dictionary with the following keys:\n            coo_train: training data in COO sparse format and as (users x items)\n             csr_train: training data in CSR sparse format and as (users x items)\n             csr_val:  validation data in CSR sparse format and as (users x items)\n    \"\"\"\n    \n    df_train, df_val = split_data(df, validation_days=validation_days)\n    coo_train = to_user_item_coo(df_train)\n    coo_val = to_user_item_coo(df_val)\n\n    csr_train = coo_train.tocsr()\n    csr_val = coo_val.tocsr()\n    \n    return {'coo_train': coo_train,\n            'csr_train': csr_train,\n            'csr_val': csr_val\n          }\n\n\ndef validate(matrices, factors=200, iterations=20, regularization=0.01, show_progress=True):\n    \"\"\" Train an ALS model with <<factors>> (embeddings dimension) \n    for <<iterations>> over matrices and validate with MAP@12\n    \"\"\"\n    coo_train, csr_train, csr_val = matrices['coo_train'], matrices['csr_train'], matrices['csr_val']\n    \n    model = implicit.als.AlternatingLeastSquares(factors=factors, \n                                                 iterations=iterations, \n                                                 regularization=regularization, \n                                                 random_state=42)\n    # factors:    計算する潜在的な要因の数\n    # iterations: データをフィットする際に使用する ALS の反復回数\n    # regularization: 使用する正則化係数(L1正則化)\n    \n    model.fit(coo_train, show_progress=show_progress)\n    #show_progress 学習の進捗を確認\n    \n    # The MAPK by implicit doesn't allow to calculate allowing repeated items, which is the case.\n    # TODO: change MAP@12 to a library that allows repeated items in prediction\n    map12 = mean_average_precision_at_k(model, csr_train, csr_val, K=12, show_progress=show_progress, num_threads=4)\n    print(f\"Factors: {factors:>3} - Iterations: {iterations:>2} - Regularization: {regularization:4.3f} ==> MAP@12: {map12:6.5f}\")\n    return map12","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:26.13636Z","iopub.execute_input":"2022-04-06T09:20:26.13655Z","iopub.status.idle":"2022-04-06T09:20:26.154801Z","shell.execute_reply.started":"2022-04-06T09:20:26.136528Z","shell.execute_reply":"2022-04-06T09:20:26.153384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:42:00.427724Z","iopub.execute_input":"2022-04-06T09:42:00.428001Z","iopub.status.idle":"2022-04-06T09:42:00.433688Z","shell.execute_reply.started":"2022-04-06T09:42:00.427969Z","shell.execute_reply":"2022-04-06T09:42:00.432279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matrices = get_val_matrices(df)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:26.156059Z","iopub.execute_input":"2022-04-06T09:20:26.156387Z","iopub.status.idle":"2022-04-06T09:20:26.428388Z","shell.execute_reply.started":"2022-04-06T09:20:26.156351Z","shell.execute_reply":"2022-04-06T09:20:26.426818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nbest_map12 = 0\nfor factors in [10, 25, 50, 75, 100]:\n    for iterations in [3, 12, 14, 15, 20]:\n        for regularization in [0.01]:\n            map12 = validate(matrices, factors, iterations, regularization, show_progress=False)\n            if map12 > best_map12:\n                best_map12 = map12\n                best_params = {'factors': factors, 'iterations': iterations, 'regularization': regularization}\n                print(f\"Best MAP@12 found. Updating: {best_params}\")","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:20:26.430738Z","iopub.execute_input":"2022-04-06T09:20:26.43102Z","iopub.status.idle":"2022-04-06T09:40:25.150546Z","shell.execute_reply.started":"2022-04-06T09:20:26.43099Z","shell.execute_reply":"2022-04-06T09:40:25.150025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del matrices","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:40:25.151362Z","iopub.execute_input":"2022-04-06T09:40:25.15152Z","iopub.status.idle":"2022-04-06T09:40:25.15857Z","shell.execute_reply.started":"2022-04-06T09:40:25.151496Z","shell.execute_reply":"2022-04-06T09:40:25.158077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Training over the full dataset","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:40:25.160784Z","iopub.execute_input":"2022-04-06T09:40:25.162731Z","iopub.status.idle":"2022-04-06T09:40:25.170084Z","shell.execute_reply.started":"2022-04-06T09:40:25.1627Z","shell.execute_reply":"2022-04-06T09:40:25.169592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coo_train = to_user_item_coo(df)\ncsr_train = coo_train.tocsr()","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:40:25.171057Z","iopub.execute_input":"2022-04-06T09:40:25.173346Z","iopub.status.idle":"2022-04-06T09:40:25.2744Z","shell.execute_reply.started":"2022-04-06T09:40:25.173313Z","shell.execute_reply":"2022-04-06T09:40:25.273477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(coo_train, factors=200, iterations=15, regularization=0.01, show_progress=True):\n    model = implicit.als.AlternatingLeastSquares(factors=factors, \n                                                 iterations=iterations, \n                                                 regularization=regularization, \n                                                 random_state=42)\n    model.fit(coo_train, show_progress=show_progress)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:40:25.275637Z","iopub.execute_input":"2022-04-06T09:40:25.275882Z","iopub.status.idle":"2022-04-06T09:40:25.282852Z","shell.execute_reply.started":"2022-04-06T09:40:25.275846Z","shell.execute_reply":"2022-04-06T09:40:25.281562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:40:25.284389Z","iopub.execute_input":"2022-04-06T09:40:25.285415Z","iopub.status.idle":"2022-04-06T09:40:25.302945Z","shell.execute_reply.started":"2022-04-06T09:40:25.285356Z","shell.execute_reply":"2022-04-06T09:40:25.302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = train(coo_train, **best_params)","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:40:25.306443Z","iopub.execute_input":"2022-04-06T09:40:25.30669Z","iopub.status.idle":"2022-04-06T09:40:53.498639Z","shell.execute_reply.started":"2022-04-06T09:40:25.306666Z","shell.execute_reply":"2022-04-06T09:40:53.497776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submit(model, csr_train, submission_name=\"submissions.csv\"):\n    preds = []\n    batch_size = 2000\n    to_generate = np.arange(len(ALL_USERS))\n    for startidx in range(0, len(to_generate), batch_size):\n        batch = to_generate[startidx : startidx + batch_size]\n        ids, scores = model.recommend(batch, csr_train[batch], N=12, filter_already_liked_items=False)\n        for i, userid in enumerate(batch):\n            customer_id = user_ids[userid]\n            user_items = ids[i]\n            article_ids = [item_ids[item_id] for item_id in user_items]\n            preds.append((customer_id, ' '.join(article_ids)))\n\n    df_preds = pd.DataFrame(preds, columns=['customer_id', 'prediction'])\n    df_preds.to_csv(submission_name, index=False)\n    \n    display(df_preds.head())\n    print(df_preds.shape)\n    \n    return df_preds","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:42:53.527743Z","iopub.execute_input":"2022-04-06T09:42:53.528014Z","iopub.status.idle":"2022-04-06T09:42:53.537617Z","shell.execute_reply.started":"2022-04-06T09:42:53.527989Z","shell.execute_reply":"2022-04-06T09:42:53.53611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_preds = submit(model, csr_train);","metadata":{"execution":{"iopub.status.busy":"2022-04-06T09:43:04.088607Z","iopub.execute_input":"2022-04-06T09:43:04.088867Z","iopub.status.idle":"2022-04-06T09:52:48.806168Z","shell.execute_reply.started":"2022-04-06T09:43:04.088843Z","shell.execute_reply":"2022-04-06T09:52:48.80431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}