{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nimport joblib\nimport gc\nimport itertools\nfrom itertools import combinations\n\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\n\nimport lightgbm as lgb\nfrom sklearn.preprocessing import LabelEncoder\nfrom hyperopt import STATUS_OK, Trials, fmin, hp, tpe\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\n\n\npd.set_option('display.width', 1000)\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\nimport warnings; warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:58:43.169809Z","iopub.execute_input":"2022-07-25T06:58:43.170274Z","iopub.status.idle":"2022-07-25T06:58:44.997913Z","shell.execute_reply.started":"2022-07-25T06:58:43.170156Z","shell.execute_reply":"2022-07-25T06:58:44.996699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_difference(data, num_features):\n    df1 = []\n    customer_ids = []\n    \n    for customer_id, df in tqdm(data.groupby(['customer_ID'])):\n        diff_df1 = df[num_features].diff(1).iloc[[-1]].values.astype(np.float16)#float32をfloat16に変えた\n        df1.append(diff_df1)\n        customer_ids.append(customer_id)\n    \n    df1 = np.concatenate(df1, axis = 0)\n    df1 = pd.DataFrame(df1, columns = [col + '_diff1' for col in df[num_features].columns])\n    df1['customer_ID'] = customer_ids\n    return df1","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:58:44.999938Z","iopub.execute_input":"2022-07-25T06:58:45.000308Z","iopub.status.idle":"2022-07-25T06:58:45.008876Z","shell.execute_reply.started":"2022-07-25T06:58:45.000274Z","shell.execute_reply":"2022-07-25T06:58:45.007589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# testデータの前処理(trainと同じ)","metadata":{}},{"cell_type":"code","source":"def read_preprocess_testdata(test_, save_index):\n    # Set feature  :特徴量の指定(customerとS_2以外のすべて): \n    features = test_.drop(['customer_ID', 'S_2'], axis = 1).columns.to_list()\n    # Set Categorical variables : カテゴリ変数の指定\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\",\n    ]\n    \n    # number of num features : 数値変数の個数(特徴量の中からカテゴリ変数で指定したものを除く)\n    num_features = [col for col in features if col not in cat_features]\n    \n    # reducing memory by casting float32 to float16\n    # floatデータのメモリ削減(もともとfloat32だったものをfloat16まで減らす)\n    print('Starting reducing memory. (num feature: cast float32 to float16)')\n    float_cols = list(test_.dtypes[test_.dtypes == 'float32'].index)\n    for col in tqdm(float_cols):\n        test_[col] = test_[col].astype(np.float16)\n    \n    # feature engineering\n    print('Starting test feature engineer...')    \n    test_num_agg = test_.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n    test_num_agg.reset_index(inplace = True)\n    test_cat_agg = test_.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n    test_cat_agg.reset_index(inplace = True)\n    \n    print('Starting reducing memory(cast 64->16) ...')\n    cols = list(test_num_agg.dtypes[test_num_agg.dtypes == 'float64'].index)\n    for col in tqdm(cols):\n        test_num_agg[col] = test_num_agg[col].astype(np.float16)\n        \n    cols = list(test_cat_agg.dtypes[test_cat_agg.dtypes == 'int64'].index)    \n    for col in tqdm(cols):\n        test_cat_agg[col] = test_cat_agg[col].astype(np.int8)\n    \n    print('Starting calculate difference...')\n    test_diff = get_difference(test_, num_features)\n    test_ = test_num_agg.merge(test_cat_agg, how = 'inner', on = 'customer_ID').merge(test_diff, how = 'inner', on = 'customer_ID')\n    \n    del test_num_agg, test_cat_agg, test_diff\n    gc.collect()\n    \n    print('Starting reducing memory. (num feature: cast float16 to float32)')\n    float_cols = list(test_.dtypes[test_.dtypes == 'float16'].index)\n    for col in tqdm(float_cols):\n        test_[col] = test_[col].astype(np.float32)\n    \n    print('Saving file ...')\n    filename = f'test_fe_{save_index}.ftr'\n    test_.to_feather(filename)\n    print(f'Finished Saving file. => {filename}')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:58:45.010146Z","iopub.execute_input":"2022-07-25T06:58:45.011164Z","iopub.status.idle":"2022-07-25T06:58:45.028470Z","shell.execute_reply.started":"2022-07-25T06:58:45.011122Z","shell.execute_reply":"2022-07-25T06:58:45.027407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing test data","metadata":{}},{"cell_type":"code","source":"test = pd.read_feather('../input/parquet-files-amexdefault-prediction/test_data.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:58:45.058785Z","iopub.execute_input":"2022-07-25T06:58:45.059642Z","iopub.status.idle":"2022-07-25T06:59:25.833969Z","shell.execute_reply.started":"2022-07-25T06:58:45.059608Z","shell.execute_reply":"2022-07-25T06:59:25.832453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_IDs = test[\"customer_ID\"].unique()\nprint(len(unique_IDs))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:59:25.836202Z","iopub.execute_input":"2022-07-25T06:59:25.836606Z","iopub.status.idle":"2022-07-25T06:59:27.785760Z","shell.execute_reply.started":"2022-07-25T06:59:25.836571Z","shell.execute_reply":"2022-07-25T06:59:27.784555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preprocessing (data is splited for reducing memory)\nread_split_num = 5\nsplited_data_IDnum = int(len(unique_IDs)/5)\nprint(splited_data_IDnum)\n\ndatanum = 0\n\nfor i in range(read_split_num):\n    print(f\"split : {i+1} / {read_split_num}\")\n    if i == read_split_num-1:\n        pick_IDs = unique_IDs[i*splited_data_IDnum:]\n    else:\n        pick_IDs = unique_IDs[i*splited_data_IDnum : (i+1)*splited_data_IDnum]\n    pick_test = test[test[\"customer_ID\"].isin(pick_IDs)]\n    datanum += len(pick_test)\n    print(f\"picked data: {len(pick_test)}. sumlen:{datanum}/{len(test)}\")\n    print(\"-- start preprocessing... --\")\n    read_preprocess_testdata(pick_test, i)\n    print(\"---\")\n    print(\" \")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:59:27.808395Z","iopub.execute_input":"2022-07-25T06:59:27.809281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}