{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import feather\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:15:24.330790Z","iopub.execute_input":"2022-07-11T10:15:24.331354Z","iopub.status.idle":"2022-07-11T10:15:24.374079Z","shell.execute_reply.started":"2022-07-11T10:15:24.331258Z","shell.execute_reply":"2022-07-11T10:15:24.373321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = feather.read_dataframe(\"../input/parquet-files-amexdefault-prediction/train_data.ftr\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:15:24.375368Z","iopub.execute_input":"2022-07-11T10:15:24.375788Z","iopub.status.idle":"2022-07-11T10:15:46.801803Z","shell.execute_reply.started":"2022-07-11T10:15:24.375757Z","shell.execute_reply":"2022-07-11T10:15:46.800846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# trainデータの中身を見る","metadata":{}},{"cell_type":"code","source":"train.shape\n# (5531451, 190)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:15:46.811345Z","iopub.execute_input":"2022-07-11T10:15:46.811701Z","iopub.status.idle":"2022-07-11T10:15:46.819444Z","shell.execute_reply.started":"2022-07-11T10:15:46.811666Z","shell.execute_reply":"2022-07-11T10:15:46.818535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n・データが大きすぎて、このままではメモリーエラーが起きる\n・大きな方向性としては二つある\n① 複数行ある行から、一行だけを残す\n→処理が軽い、学習時の精度が悪い\n② 全ての行を集約する\n→処理が軽い、学習時の精度が良い\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:15:46.820607Z","iopub.execute_input":"2022-07-11T10:15:46.821015Z","iopub.status.idle":"2022-07-11T10:15:46.833826Z","shell.execute_reply.started":"2022-07-11T10:15:46.820962Z","shell.execute_reply":"2022-07-11T10:15:46.833058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ① 複数行ある行から、一行だけを残す\n","metadata":{}},{"cell_type":"code","source":"\"\"\"\n一人のcustomer_idに関して、複数行のデータが入っている\n\"\"\"\ncust_one=train[\"customer_ID\"].unique()[0]\ntrain[train[\"customer_ID\"]==cust_one]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:15:46.835333Z","iopub.execute_input":"2022-07-11T10:15:46.835949Z","iopub.status.idle":"2022-07-11T10:15:48.138077Z","shell.execute_reply.started":"2022-07-11T10:15:46.835916Z","shell.execute_reply":"2022-07-11T10:15:48.137227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"drop_duplicatesにより、データ容量を削減することが可能に\"\"\"\ntrain.drop_duplicates(subset=['customer_ID'], keep='last')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:15:48.139511Z","iopub.execute_input":"2022-07-11T10:15:48.139890Z","iopub.status.idle":"2022-07-11T10:15:50.289811Z","shell.execute_reply.started":"2022-07-11T10:15:48.139860Z","shell.execute_reply":"2022-07-11T10:15:50.288955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ② 全ての行を集約する\n","metadata":{}},{"cell_type":"code","source":"cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\",\n    ]\nnum_features = [col for col in train.columns if col not in cat_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:15:50.291322Z","iopub.execute_input":"2022-07-11T10:15:50.291875Z","iopub.status.idle":"2022-07-11T10:15:50.297638Z","shell.execute_reply.started":"2022-07-11T10:15:50.291827Z","shell.execute_reply":"2022-07-11T10:15:50.296690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ②のサンプル(K=1のみ作成)","metadata":{}},{"cell_type":"code","source":"\"\"\"customer_idを区切って、バッチ処理を行う\"\"\"\nNUM_SPLIT = 10\ncount = 0\nstart_row = 0\npred_df = pd.DataFrame()\nunique_id = train['customer_ID'].unique().tolist()\nnum_split_id = len(unique_id) // NUM_SPLIT\n\n\nk=1\nprint('Current split: %s' % k)\nend_row = start_row + num_split_id\n# 該当のデータを取得する\nif k < NUM_SPLIT:\n    cur_id = unique_id[start_row : end_row]\nelse:\n    cur_id = unique_id[start_row: ]\n\ntrain_num_agg = train[train[\"customer_ID\"].isin(cur_id)].groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:16:03.996687Z","iopub.execute_input":"2022-07-11T10:16:03.997121Z","iopub.status.idle":"2022-07-11T10:16:14.076067Z","shell.execute_reply.started":"2022-07-11T10:16:03.997082Z","shell.execute_reply":"2022-07-11T10:16:14.075053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"集約された特徴量を作成\"\"\"\ntrain_num_agg.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:16:23.298951Z","iopub.execute_input":"2022-07-11T10:16:23.299331Z","iopub.status.idle":"2022-07-11T10:16:23.331973Z","shell.execute_reply.started":"2022-07-11T10:16:23.299300Z","shell.execute_reply":"2022-07-11T10:16:23.331331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"csvファイルがoutputに保存される\"\"\"\ntrain_num_agg.to_csv(f\"train_num_agg{k}.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T10:18:08.381921Z","iopub.execute_input":"2022-07-11T10:18:08.382808Z","iopub.status.idle":"2022-07-11T10:18:47.130777Z","shell.execute_reply.started":"2022-07-11T10:18:08.382757Z","shell.execute_reply":"2022-07-11T10:18:47.129908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"\"\"customer_idを区切って、バッチ処理を行う\"\"\"\n# NUM_SPLIT = 10\n# count = 0\n# start_row = 0\n# pred_df = pd.DataFrame()\n# unique_id = train['customer_ID'].unique().tolist()\n# num_split_id = len(unique_id) // NUM_SPLIT\n# for k in range(1, NUM_SPLIT + 1):\n#     print('Current split: %s' % k)\n#     end_row = start_row + num_split_id\n#     # 該当のデータを取得する\n#     if k < NUM_SPLIT:\n#         cur_id = unique_id[start_row : end_row]\n#     else:\n#         cur_id = unique_id[start_row: ]\n        \n#     train_num_agg = train[train[\"customer_ID\"].isin(cur_id)].groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n#     train_num_agg.to_csv(f\"train_num_agg{k}.csv\")\n    \n#     start_row = end_row","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}