{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Preparation","metadata":{}},{"cell_type":"markdown","source":"### Load packages","metadata":{}},{"cell_type":"code","source":"import time\nimport gc\nimport pickle\n\nimport pandas as pd\nimport numpy as np\n\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T05:10:41.9706Z","iopub.execute_input":"2022-07-07T05:10:41.971931Z","iopub.status.idle":"2022-07-07T05:10:42.485294Z","shell.execute_reply.started":"2022-07-07T05:10:41.971788Z","shell.execute_reply":"2022-07-07T05:10:42.483688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configurations","metadata":{}},{"cell_type":"code","source":"pd.options.display.max_rows = 300\npd.options.display.max_seq_items = 300\nDEBUG = False\nNROWS_DEBUG = 1000\nSAVE_PREPROCESSED_DATA=True\ndata_version=int(time.time())\nprint('Data version: {}'.format(data_version))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:10:42.487629Z","iopub.execute_input":"2022-07-07T05:10:42.488498Z","iopub.status.idle":"2022-07-07T05:10:42.496843Z","shell.execute_reply.started":"2022-07-07T05:10:42.488447Z","shell.execute_reply":"2022-07-07T05:10:42.495354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Utility Functions","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:10:42.49933Z","iopub.execute_input":"2022-07-07T05:10:42.500628Z","iopub.status.idle":"2022-07-07T05:10:42.517464Z","shell.execute_reply.started":"2022-07-07T05:10:42.500554Z","shell.execute_reply":"2022-07-07T05:10:42.516274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocess","metadata":{}},{"cell_type":"markdown","source":"### Load\nLoad aggregrate data pickle","metadata":{}},{"cell_type":"code","source":"print('Load data ...')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif DEBUG:\n    train = pd.read_pickle(\"../input/amex-agg-data-pickle/train_agg.pkl\", compression=\"gzip\")\n    train = train.head(NROWS_DEBUG)\n    test = pd.read_pickle(\"../input/amex-agg-data-pickle/test_agg.pkl\", compression=\"gzip\")\n    test = test.head(NROWS_DEBUG)\nelse:\n    train = pd.read_pickle(\"../input/amex-agg-data-pickle/train_agg.pkl\", compression=\"gzip\")\n    test = pd.read_pickle(\"../input/amex-agg-data-pickle/test_agg.pkl\", compression=\"gzip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:10:42.520646Z","iopub.execute_input":"2022-07-07T05:10:42.522075Z","iopub.status.idle":"2022-07-07T05:11:18.685619Z","shell.execute_reply.started":"2022-07-07T05:10:42.522002Z","shell.execute_reply":"2022-07-07T05:11:18.684117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = reduce_mem_usage(train)\ntest = reduce_mem_usage(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:18.687263Z","iopub.execute_input":"2022-07-07T05:11:18.688305Z","iopub.status.idle":"2022-07-07T05:11:19.724003Z","shell.execute_reply.started":"2022-07-07T05:11:18.688261Z","shell.execute_reply":"2022-07-07T05:11:19.722862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Label Encoder / One Hot Encoder","metadata":{}},{"cell_type":"code","source":"print('Label and One Hot Encoder ...')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_63\",\n    \"D_64\",\n    \"D_66\",\n    \"D_68\"\n]\ncat_features = [f\"{cf}_last\" for cf in cat_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:19.725878Z","iopub.execute_input":"2022-07-07T05:11:19.726306Z","iopub.status.idle":"2022-07-07T05:11:19.732254Z","shell.execute_reply.started":"2022-07-07T05:11:19.726271Z","shell.execute_reply":"2022-07-07T05:11:19.730882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nle = LabelEncoder()\nohc = OneHotEncoder()\n\nfor col in cat_features:\n\n    # Train\n    # Integer encode the string categories\n    train_col = le.fit_transform(train[col])\n    # One hot encode the data--this returns a sparse array\n    new_dat = ohc.fit_transform(train_col.reshape(-1,1))\n    # Create unique column names\n    n_cols = new_dat.shape[1]\n    col_names = ['_cat_'.join([col, str(x)]) for x in range(n_cols)]\n    # Create the new dataframe\n    new_df = pd.DataFrame(new_dat.toarray(),\n                          index=train.index, \n                          columns=col_names)\n    # Append the new data to the dataframe\n    train = pd.concat([train, new_df], axis=1)\n    # Remove the original column from the dataframe\n    train = train.drop(col, axis=1)\n    \n    \n    # Test\n    test_col = le.transform(test[col])\n    new_dat = ohc.transform(test_col.reshape(-1, 1))\n    new_df = pd.DataFrame(new_dat.toarray(),\n                         index=test.index,\n                         columns=col_names)\n    test = pd.concat([test, new_df], axis=1)\n    test = test.drop(col, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:19.733948Z","iopub.execute_input":"2022-07-07T05:11:19.734535Z","iopub.status.idle":"2022-07-07T05:11:20.498615Z","shell.execute_reply.started":"2022-07-07T05:11:19.734483Z","shell.execute_reply":"2022-07-07T05:11:20.497253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = reduce_mem_usage(train)\ntest = reduce_mem_usage(test)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:20.499615Z","iopub.execute_input":"2022-07-07T05:11:20.499939Z","iopub.status.idle":"2022-07-07T05:11:21.669423Z","shell.execute_reply.started":"2022-07-07T05:11:20.49991Z","shell.execute_reply":"2022-07-07T05:11:21.668129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill NA","metadata":{}},{"cell_type":"code","source":"print('Fill NA ...')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nna_percentages = {}\nvalue_fillna = {}\nnrows = train.shape[0]\n\nfor col in train.columns:\n    if col != 'target':\n        num_na = train[col].isna().sum()\n        if num_na > 0:\n            na_percentages[col] = num_na / nrows * 100\n            value_fillna[col] = train[col].mean()\n            \n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:21.671246Z","iopub.execute_input":"2022-07-07T05:11:21.671588Z","iopub.status.idle":"2022-07-07T05:11:21.859848Z","shell.execute_reply.started":"2022-07-07T05:11:21.671559Z","shell.execute_reply":"2022-07-07T05:11:21.858464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor col, val in value_fillna.items():\n    train[col].fillna(value=val, inplace=True)\n    test[col].fillna(value=val, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:21.862645Z","iopub.execute_input":"2022-07-07T05:11:21.863384Z","iopub.status.idle":"2022-07-07T05:11:21.998565Z","shell.execute_reply.started":"2022-07-07T05:11:21.863339Z","shell.execute_reply":"2022-07-07T05:11:21.997167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = reduce_mem_usage(train)\ntest = reduce_mem_usage(test)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:22.000744Z","iopub.execute_input":"2022-07-07T05:11:22.001598Z","iopub.status.idle":"2022-07-07T05:11:22.987968Z","shell.execute_reply.started":"2022-07-07T05:11:22.001542Z","shell.execute_reply":"2022-07-07T05:11:22.986674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save Preprocessed Data","metadata":{}},{"cell_type":"code","source":"print('Save Processed Data ...')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif SAVE_PREPROCESSED_DATA:\n    train.to_pickle(\"train_agg_data_{}.pkl\".format(data_version), compression=\"gzip\")\n    test.to_pickle(\"test_agg_data_{}.pkl\".format(data_version), compression=\"gzip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:22.989284Z","iopub.execute_input":"2022-07-07T05:11:22.989619Z","iopub.status.idle":"2022-07-07T05:11:32.499841Z","shell.execute_reply.started":"2022-07-07T05:11:22.989589Z","shell.execute_reply":"2022-07-07T05:11:32.498538Z"},"trusted":true},"execution_count":null,"outputs":[]}]}