{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport dask.dataframe as dd\nimport os, gc\nfrom sklearn.impute import SimpleImputer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-16T20:24:48.359191Z","iopub.execute_input":"2022-06-16T20:24:48.359664Z","iopub.status.idle":"2022-06-16T20:24:50.177634Z","shell.execute_reply.started":"2022-06-16T20:24:48.359565Z","shell.execute_reply":"2022-06-16T20:24:50.176565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file_cpu(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = pd.read_feather(path, columns=usecols)\n    else: df = pd.read_feather(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n#   df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = pd.to_datetime(df.S_2)\n    # CREATE OVERALL ROW MISS VALUE\n    features = [x for x in df.columns.values if x not in ['customer_ID', 'target']]\n    #df['n_missing'] = df[features].isna().sum(axis=1)\n    # FILL NAN\n#     features_num = [x for x in df._get_numeric_data().columns.values if x not in ['customer_ID', 'target']]\n#     df = df[features_num].fillna(NAN_VALUE) \n    # KEEP ONLY FINAL CUSTOMER ID UNTIL FUTURE TIME SERIES WORK BEGINS\n    df_out = df.groupby(['customer_ID']).nth(-1).reset_index(drop=False)\n    print('shape of data:', df_out.shape)\n    del df\n    _ = gc.collect()\n    return df_out","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:49:13.839029Z","iopub.execute_input":"2022-06-16T21:49:13.840073Z","iopub.status.idle":"2022-06-16T21:49:13.850348Z","shell.execute_reply.started":"2022-06-16T21:49:13.840005Z","shell.execute_reply":"2022-06-16T21:49:13.849354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading train data...')\nTRAIN_PATH = '../input/amexfeather/train_data.ftr'\ntrain_df = read_file_cpu(path = TRAIN_PATH)\n\nprint('Reading test data...')\nTEST_PATH = '../input/amexfeather/test_data.ftr'\ntest_df = read_file_cpu(path = TEST_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:49:18.355321Z","iopub.execute_input":"2022-06-16T21:49:18.355972Z","iopub.status.idle":"2022-06-16T21:50:48.284043Z","shell.execute_reply.started":"2022-06-16T21:49:18.355932Z","shell.execute_reply":"2022-06-16T21:50:48.282965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:50:57.168126Z","iopub.execute_input":"2022-06-16T21:50:57.168565Z","iopub.status.idle":"2022-06-16T21:50:57.197886Z","shell.execute_reply.started":"2022-06-16T21:50:57.168530Z","shell.execute_reply":"2022-06-16T21:50:57.196973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = pd.Series(data=[1.234567,0.000078,np.nan,1.356876])\nx = x.astype('float64')\nx.ffill()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:48:14.581862Z","iopub.execute_input":"2022-06-16T20:48:14.582278Z","iopub.status.idle":"2022-06-16T20:48:14.590944Z","shell.execute_reply.started":"2022-06-16T20:48:14.582247Z","shell.execute_reply":"2022-06-16T20:48:14.590263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_df['P_2'][190:198]\ny","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:45:50.915029Z","iopub.execute_input":"2022-06-16T20:45:50.915445Z","iopub.status.idle":"2022-06-16T20:45:50.922979Z","shell.execute_reply.started":"2022-06-16T20:45:50.915413Z","shell.execute_reply":"2022-06-16T20:45:50.922125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_64 = np.array(train_df['D_64'])\n\nfor i in range(len(d_64)):\n    if d_64[i] == '':\n        d_64[i] = 'X'\n        \ntrain_df['D_64'] = d_64","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:48:55.003394Z","iopub.execute_input":"2022-06-16T20:48:55.003860Z","iopub.status.idle":"2022-06-16T20:48:55.102392Z","shell.execute_reply.started":"2022-06-16T20:48:55.003821Z","shell.execute_reply":"2022-06-16T20:48:55.101254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = train_df.columns\nfor i in range(len(cols)):\n    if train_df[cols[i]].isna().sum() >= (train_df.shape[0] * 0.30):\n        train_df = train_df.drop(cols[i],axis=1)\n            \ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:25.880896Z","iopub.execute_input":"2022-06-16T20:49:25.881338Z","iopub.status.idle":"2022-06-16T20:49:35.259967Z","shell.execute_reply.started":"2022-06-16T20:49:25.881301Z","shell.execute_reply":"2022-06-16T20:49:35.258928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For some reason the ffill function does not work for dtypes float16 therefore they need to be converted to float64 dtypes","metadata":{}},{"cell_type":"code","source":"new_cols = train_df.columns\nfor i in range(len(new_cols)):\n    if train_df[new_cols[i]].dtype.name == 'float16':\n        train_df[new_cols[i]] = train_df[new_cols[i]].astype('float64')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:51:26.721367Z","iopub.execute_input":"2022-06-16T20:51:26.722191Z","iopub.status.idle":"2022-06-16T20:51:30.670586Z","shell.execute_reply.started":"2022-06-16T20:51:26.722143Z","shell.execute_reply":"2022-06-16T20:51:30.669738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:52:07.607412Z","iopub.execute_input":"2022-06-16T20:52:07.607882Z","iopub.status.idle":"2022-06-16T20:52:07.630537Z","shell.execute_reply.started":"2022-06-16T20:52:07.607843Z","shell.execute_reply":"2022-06-16T20:52:07.629456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.ffill()\ntrain_df = train_df.bfill()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:52:53.501229Z","iopub.execute_input":"2022-06-16T20:52:53.501586Z","iopub.status.idle":"2022-06-16T20:52:53.885654Z","shell.execute_reply.started":"2022-06-16T20:52:53.501554Z","shell.execute_reply":"2022-06-16T20:52:53.884654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_cols = train_df.columns\nfor i in range(len(new_cols)):\n    if train_df[new_cols[i]].dtype.name == 'float64':\n        train_df[new_cols[i]] = train_df[new_cols[i]].astype('float16')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:54:07.819029Z","iopub.execute_input":"2022-06-16T20:54:07.820523Z","iopub.status.idle":"2022-06-16T20:54:19.684984Z","shell.execute_reply.started":"2022-06-16T20:54:07.820452Z","shell.execute_reply":"2022-06-16T20:54:19.683789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:54:56.024015Z","iopub.execute_input":"2022-06-16T20:54:56.024495Z","iopub.status.idle":"2022-06-16T20:54:56.047857Z","shell.execute_reply.started":"2022-06-16T20:54:56.024455Z","shell.execute_reply":"2022-06-16T20:54:56.046720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_feather('train_data')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:07:16.259102Z","iopub.execute_input":"2022-06-16T21:07:16.259463Z","iopub.status.idle":"2022-06-16T21:07:16.284132Z","shell.execute_reply.started":"2022-06-16T21:07:16.259432Z","shell.execute_reply":"2022-06-16T21:07:16.283067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = train_df.columns\nlen(cols)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:08:20.853510Z","iopub.execute_input":"2022-06-16T21:08:20.853928Z","iopub.status.idle":"2022-06-16T21:08:20.859743Z","shell.execute_reply.started":"2022-06-16T21:08:20.853889Z","shell.execute_reply":"2022-06-16T21:08:20.858825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_cols = test_df.columns\nfor i in range(len(test_cols)):\n    if test_cols[i] not in cols:\n        test_df = test_df.drop(test_cols[i],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:10:36.160431Z","iopub.execute_input":"2022-06-16T21:10:36.160757Z","iopub.status.idle":"2022-06-16T21:10:57.436352Z","shell.execute_reply.started":"2022-06-16T21:10:36.160730Z","shell.execute_reply":"2022-06-16T21:10:57.435406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:14:12.344468Z","iopub.execute_input":"2022-06-16T21:14:12.344961Z","iopub.status.idle":"2022-06-16T21:14:12.365614Z","shell.execute_reply.started":"2022-06-16T21:14:12.344915Z","shell.execute_reply":"2022-06-16T21:14:12.364900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_64_t = np.array(test_df['D_64'])\n\nfor i in range(len(d_64_t)):\n    if d_64_t[i] == '':\n        d_64_t[i] = 'X'\n        \ntest_df['D_64'] = d_64_t","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:16:03.051549Z","iopub.execute_input":"2022-06-16T21:16:03.051988Z","iopub.status.idle":"2022-06-16T21:16:03.192354Z","shell.execute_reply.started":"2022-06-16T21:16:03.051956Z","shell.execute_reply":"2022-06-16T21:16:03.191499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_cols_t = test_df.columns\nfor i in range(len(new_cols_t)):\n    if test_df[new_cols_t[i]].dtype.name == 'float16':\n        test_df[new_cols_t[i]] = test_df[new_cols_t[i]].astype('float64')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:16:48.095811Z","iopub.execute_input":"2022-06-16T21:16:48.096250Z","iopub.status.idle":"2022-06-16T21:16:56.843637Z","shell.execute_reply.started":"2022-06-16T21:16:48.096216Z","shell.execute_reply":"2022-06-16T21:16:56.842580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.ffill()\ntest_df = test_df.bfill()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:17:16.888395Z","iopub.execute_input":"2022-06-16T21:17:16.889488Z","iopub.status.idle":"2022-06-16T21:17:20.417451Z","shell.execute_reply.started":"2022-06-16T21:17:16.889444Z","shell.execute_reply":"2022-06-16T21:17:20.416232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_cols_t = test_df.columns\nfor i in range(len(new_cols_t)):\n    if test_df[new_cols_t[i]].dtype.name == 'float64':\n        test_df[new_cols_t[i]] = test_df[new_cols_t[i]].astype('float16')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:18:28.132104Z","iopub.execute_input":"2022-06-16T21:18:28.133474Z","iopub.status.idle":"2022-06-16T21:19:04.163881Z","shell.execute_reply.started":"2022-06-16T21:18:28.133406Z","shell.execute_reply":"2022-06-16T21:19:04.162792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_feather('test_data')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T21:19:59.143832Z","iopub.execute_input":"2022-06-16T21:19:59.144377Z","iopub.status.idle":"2022-06-16T21:19:59.966831Z","shell.execute_reply.started":"2022-06-16T21:19:59.144325Z","shell.execute_reply":"2022-06-16T21:19:59.965999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}