{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Create popular items list!!!","metadata":{}},{"cell_type":"markdown","source":"購入された回数が多い商品はやっぱり人気が高いと思います。\n\nなので、学習用データとテスト用データから実際に注文された回数が多い商品のリストを作ります。","metadata":{}},{"cell_type":"markdown","source":"# 1.ライブラリのインポート","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport time\nimport gc\nimport copy\nimport glob\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:01:16.141435Z","iopub.execute_input":"2022-12-29T08:01:16.141983Z","iopub.status.idle":"2022-12-29T08:01:16.148342Z","shell.execute_reply.started":"2022-12-29T08:01:16.141942Z","shell.execute_reply":"2022-12-29T08:01:16.147086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.リストの作成","metadata":{}},{"cell_type":"code","source":"%%time\n# 与えられたパスのファイルを読み込む関数。\ndef read_file(f):\n    return pd.DataFrame(data_cache[f])\n\n\n# ファイルの容量をできるだけ小さくする関数。\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = ((df.ts // 1000)).astype('int32')\n    df['type'] = df['type'].map(type_list).astype('int8')\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:01:16.151226Z","iopub.execute_input":"2022-12-29T08:01:16.151692Z","iopub.status.idle":"2022-12-29T08:01:16.165195Z","shell.execute_reply.started":"2022-12-29T08:01:16.151641Z","shell.execute_reply":"2022-12-29T08:01:16.164180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cache = {}\ntype_list = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:01:16.166429Z","iopub.execute_input":"2022-12-29T08:01:16.166942Z","iopub.status.idle":"2022-12-29T08:01:16.193150Z","shell.execute_reply.started":"2022-12-29T08:01:16.166911Z","shell.execute_reply":"2022-12-29T08:01:16.191909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f in files:\n    data_cache[f] = read_file_to_cache(f)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:01:16.195887Z","iopub.execute_input":"2022-12-29T08:01:16.196390Z","iopub.status.idle":"2022-12-29T08:02:09.669715Z","shell.execute_reply.started":"2022-12-29T08:01:16.196347Z","shell.execute_reply":"2022-12-29T08:02:09.668663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"READ_CT = 5\nCHUNK = int( np.ceil( len(files)/6 ))\nprint(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK}.')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:02:09.671080Z","iopub.execute_input":"2022-12-29T08:02:09.672150Z","iopub.status.idle":"2022-12-29T08:02:09.678427Z","shell.execute_reply.started":"2022-12-29T08:02:09.672100Z","shell.execute_reply":"2022-12-29T08:02:09.676936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# メモリエラーにならない最小の階数に分けて処理をする。\nDISK_PIECES = 1\nSIZE = 1.86e6/DISK_PIECES\n\nfor PART in range(DISK_PIECES):\n    print()\n    print('### DISK PART', PART+1)\n    \n    for j in range(6):\n        a = j*CHUNK\n        b = min( (j+1)*CHUNK, len(files) )\n        print(f'Processing files {a} thru {b-1} in groups of {j+1} / 6.')\n        \n        for k in range(a,b,READ_CT):\n            df = [read_file(files[k])]\n            for i in range(1,READ_CT): \n                if k+i<b:\n                    df.append(read_file(files[k+i]))\n                    \n            # リスト型として保存していたデータフレームをデータフレーム型として結合する。\n            df = pd.concat(df, ignore_index=True,axis=0)\n            \n            # セッションは昇順、タイムスタンプは降順で並び替える。\n            df = df.sort_values(['session','ts'],ascending=[True,False])\n            df = df.reset_index(drop=True)\n            \n            if k == a:\n                tmp_inner = df\n            else:\n                tmp_inner = pd.concat([tmp_inner, df], axis=0)\n                \n        if a == 0:\n            tmp = tmp_inner\n        else:\n            tmp = pd.concat([tmp, tmp_inner], axis=0)\n        del tmp_inner, df\n        gc.collect()\n        \n    tmp = tmp.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:02:09.680600Z","iopub.execute_input":"2022-12-29T08:02:09.681169Z","iopub.status.idle":"2022-12-29T08:05:00.573871Z","shell.execute_reply.started":"2022-12-29T08:02:09.681108Z","shell.execute_reply":"2022-12-29T08:05:00.572308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ordered_list = tmp.copy()\n\n# 実際に注文された商品だけを取り出す。\nordered_list = ordered_list.loc[(ordered_list['type'] == 2), ]\n\n# 不要な行(sessionやts)は削除する。\nordered_list.drop(['session', 'ts'], axis=1, inplace=True)\n\n# 注文された商品のみを取り出したデータフレームから、aidごとに出現回数を数える。\npopular_ranking = ordered_list.groupby('aid').count()\n\ndel tmp, ordered_list, data_cache, files\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:07:54.197814Z","iopub.execute_input":"2022-12-29T08:07:54.198326Z","iopub.status.idle":"2022-12-29T08:07:57.785693Z","shell.execute_reply.started":"2022-12-29T08:07:54.198292Z","shell.execute_reply":"2022-12-29T08:07:57.784485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# typeに出現回数が入っているので、typeを降順に並べ替えて、上位20個を取り出す。\npopular_items_top20 = popular_ranking.sort_values('type', ascending=False).head(20)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:08:01.994449Z","iopub.execute_input":"2022-12-29T08:08:01.995270Z","iopub.status.idle":"2022-12-29T08:08:02.066161Z","shell.execute_reply.started":"2022-12-29T08:08:01.995222Z","shell.execute_reply":"2022-12-29T08:08:02.064848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"popular_items_top20.to_parquet('popular_items_top20.parquet')\ndel popular_items_top20\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T08:05:41.907614Z","iopub.status.idle":"2022-12-29T08:05:41.908633Z","shell.execute_reply.started":"2022-12-29T08:05:41.908168Z","shell.execute_reply":"2022-12-29T08:05:41.908199Z"},"trusted":true},"execution_count":null,"outputs":[]}]}