{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Pair items ranking!!!","metadata":{}},{"cell_type":"markdown","source":"ある商品をクリックした後、ユーザーはどの商品を購入するでしょう。\n\n多くの人が購入している商品はやはり人気が高いです。\n\nなので、すべてのユーザーの直近の行動から次に買われる、もしくはカートに入れられる可能性が高い商品を取り出してみます。","metadata":{}},{"cell_type":"markdown","source":"# 1. ライブラリのインポート","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport datetime\nimport time\nimport pyarrow as pa\nimport pyarrow.parquet as pq\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nimport gc\nimport glob\nimport numpy as np\nimport copy","metadata":{"execution":{"iopub.status.busy":"2022-12-29T06:58:11.261817Z","iopub.execute_input":"2022-12-29T06:58:11.262307Z","iopub.status.idle":"2022-12-29T06:58:12.013232Z","shell.execute_reply.started":"2022-12-29T06:58:11.262205Z","shell.execute_reply":"2022-12-29T06:58:12.011847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.行列の作成","metadata":{}},{"cell_type":"code","source":"%%time\n# 与えられたパスのファイルを読み込む関数\ndef read_file(f):\n    return pd.DataFrame(data_cache[f])\n\n\n# ファイルのの容量をできるだけ小さくする関数。\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = ((df.ts // 1000)).astype('int32')\n    df['type'] = df['type'].map(type_list).astype('int8')\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-12-29T06:58:12.015879Z","iopub.execute_input":"2022-12-29T06:58:12.016291Z","iopub.status.idle":"2022-12-29T06:58:12.026161Z","shell.execute_reply.started":"2022-12-29T06:58:12.016253Z","shell.execute_reply":"2022-12-29T06:58:12.024939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cache = {}\ntype_list = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T06:58:12.028304Z","iopub.execute_input":"2022-12-29T06:58:12.028949Z","iopub.status.idle":"2022-12-29T06:58:12.056073Z","shell.execute_reply.started":"2022-12-29T06:58:12.028906Z","shell.execute_reply":"2022-12-29T06:58:12.054683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f in files:\n    data_cache[f] = read_file_to_cache(f)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T06:58:12.058620Z","iopub.execute_input":"2022-12-29T06:58:12.059004Z","iopub.status.idle":"2022-12-29T06:59:08.698063Z","shell.execute_reply.started":"2022-12-29T06:58:12.058972Z","shell.execute_reply":"2022-12-29T06:59:08.696585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"READ_CT = 5\nCHUNK = int( np.ceil( len(files)/6 ))\nprint(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK}.')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T06:59:08.699839Z","iopub.execute_input":"2022-12-29T06:59:08.700310Z","iopub.status.idle":"2022-12-29T06:59:08.708096Z","shell.execute_reply.started":"2022-12-29T06:59:08.700259Z","shell.execute_reply":"2022-12-29T06:59:08.706811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# メモリエラーにならない最小の回数に分けて処理をする。\nDISK_PIECES = 5\nSIZE = 1.86e6/DISK_PIECES\n\n# 買ってもらえる商品をレコメンドしないといけない。\n# その点でいうと、既に買われた商品をもう一度買う確率よりも、カートに入れた商品を買う確率のほうが高い。\n# なので、カートに入れた商品に対する重みを最も大きくしている。\n\ntype_weight = {0:1, 1:6, 2:3}\n\n\nfor PART in range(DISK_PIECES):\n    print()\n    print('### DISK PART', PART+1)\n    \n    for j in range(6):\n        a = j*CHUNK\n        b = min( (j+1)*CHUNK, len(files) )\n        print(f'Processing files {a} thru {b-1} in groups of {j+1} / 6.')\n        \n        \n        for k in range(a,b,READ_CT):\n            df = [read_file(files[k])]\n            for i in range(1,READ_CT): \n                if k+i<b:\n                    df.append(read_file(files[k+i]))\n                    \n            # リスト型として保存していたデータフレームをデータフレーム型として結合する。\n            df = pd.concat(df, ignore_index=True,axis=0)\n            \n            # セッションは昇順、タイムスタンプは降順で並び替える。\n            df = df.sort_values(['session','ts'],ascending=[True,False])\n            df = df.reset_index(drop=True)\n            \n            # セッションごとに通し番号を付ける。\n            df['n'] = df.groupby('session').cumcount()\n            \n            # すべての商品に対して計算しているとメモリエラーになるので、直近30件のみを用いて行列を作成する。\n            df = df.loc[df.n<30].drop('n', axis=1)\n            df = df.merge(df, on='session')\n            \n            # このようにmergeをすると、aid_xとaid_yが同じ行が作成されてしまう。\n            # その行を削除すると同時に、次の行動までの時間が1日以上空いている場合も削除する。\n            # おそらく、1日以上空いてしまうと、まったく違うものを見ている可能性が高くなるから。\n            df = df.loc[((df.ts_x - df.ts_y).abs() < 24 * 60 * 60) & (df.aid_x != df.aid_y)]\n            \n            # メモリエラーにならないようにaid_xが指定の値の間のものを順番に処理していく。\n            df = df.loc[(PART * SIZE <= df.aid_x) & (df.aid_x < (PART + 1) * SIZE)]\n            \n            # 重複行の削除。\n            df = df[['session', 'aid_x', 'aid_y', 'type_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            \n            # aid_yの行動に対して重みを付けていく。\n            df['wgt'] = df.type_y.map(type_weight)\n            \n            # 商品ごとの関係性を表す行列を作っているので、不要な行(sessionやts)は削除する。\n            df = df[['aid_x', 'aid_y', 'wgt']]\n            \n            # 使用メモリを抑えるために型変換。\n            df.wgt = df.wgt.astype('float32')\n            \n            # 今、同じaid_xとaid_yのペアが複数存在している可能性がある。\n            # なので、同じペアならすべての重みを足し合わせる。\n            df = df.groupby(['aid_x', 'aid_y']).wgt.sum()\n            \n            if k == a:\n                tmp2 = df\n            else:\n                tmp2 = tmp2.add(df, fill_value=0)\n        if a == 0:\n            tmp = tmp2\n        else:\n            tmp = tmp.add(tmp2, fill_value=0)\n        \n        del tmp2, df\n        gc.collect()\n    \n    tmp = tmp.reset_index()\n    \n    #aid_xは昇順、wgtは降順で並び替えて、インデックスを振りなおす。\n    tmp = tmp.sort_values(['aid_x', 'wgt'], ascending=[True, False])\n    tmp = tmp.reset_index(drop=True)\n    \n    # ある商品aid_xごとに通し番号を付ける。\n    # ある商品aid_xに対し、スコアの高い順に上位20個のみを保存する。\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n < 20].drop('n', axis=1)\n    tmp.to_parquet(f'top_20_item×item_{PART}.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T07:02:30.912215Z","iopub.execute_input":"2022-12-29T07:02:30.912676Z","iopub.status.idle":"2022-12-29T07:02:37.465518Z","shell.execute_reply.started":"2022-12-29T07:02:30.912638Z","shell.execute_reply":"2022-12-29T07:02:37.464275Z"},"trusted":true},"execution_count":null,"outputs":[]}]}