{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport pandas as pd \nimport numpy as np\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.impute import SimpleImputer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-05T04:20:45.004826Z","iopub.execute_input":"2023-02-05T04:20:45.005536Z","iopub.status.idle":"2023-02-05T04:20:45.792410Z","shell.execute_reply.started":"2023-02-05T04:20:45.005403Z","shell.execute_reply":"2023-02-05T04:20:45.791350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Helper functions to impute, encode, and generate features/targets","metadata":{}},{"cell_type":"code","source":"def one_hot_encode_categorical(cat_features, cat_names):\n    \"\"\"\n    One-hot encodes categorical features using scikit-learn OneHotEncoder\n\n    Parameters\n    ----------\n    cat_features : pd.DataFrame\n        DataFrame, with index, that has only the categorical columns to one-hot encode\n    cat_names : list\n        list of categorical column names \n\n    Returns\n    -------\n    pd.DataFrame\n        DataFrame that holds each of the one-hot encoded columns \n    \"\"\"\n    \n    enc = OneHotEncoder(sparse=False)\n    encoded_df = pd.DataFrame(enc.fit_transform(cat_features), columns=enc.get_feature_names(cat_names), index=cat_features.index)\n    return encoded_df","metadata":{"execution":{"iopub.status.busy":"2023-02-05T04:20:45.794463Z","iopub.execute_input":"2023-02-05T04:20:45.795093Z","iopub.status.idle":"2023-02-05T04:20:45.800758Z","shell.execute_reply.started":"2023-02-05T04:20:45.795059Z","shell.execute_reply":"2023-02-05T04:20:45.799530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def simple_impute_numerical(numeric_features, numeric_names):\n    \"\"\"\n    Imputes numerical columns with scikit-learn SimpleImputer()\n\n    Parameters\n    ----------\n    numeric_features : pd.DataFrame\n        DataFrame, with index, that has only the numerical columns to impute\n    numeric_features : list\n        list of numerical column names \n\n    Returns\n    -------\n    pd.DataFrame\n        DataFrame that holds each of the imputed numerical columns\n    \"\"\"\n    \n    # current numeric columns are float16, and they will not work when computing mean()\n    # need to convert to float32\n    for column in numeric_features.columns:\n        numeric_features[column] = numeric_features[column].astype(np.float32)\n\n    # impute columns using the mean\n    imp_mean = SimpleImputer(missing_values=np.nan, strategy='mean')\n    numeric_df = pd.DataFrame(imp_mean.fit_transform(numeric_features), columns=numeric_names, index=numeric_features.index)\n    \n    # convert back to float16 for lighter load\n    for column in numeric_df.columns:\n        numeric_df[column] = numeric_df[column].astype(np.float16)\n    \n    return numeric_df","metadata":{"execution":{"iopub.status.busy":"2023-02-05T04:20:45.802776Z","iopub.execute_input":"2023-02-05T04:20:45.803372Z","iopub.status.idle":"2023-02-05T04:20:45.814676Z","shell.execute_reply.started":"2023-02-05T04:20:45.803310Z","shell.execute_reply":"2023-02-05T04:20:45.813370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_x_y(df_file_path, test=False):\n    \"\"\"\n    Returns the features (X) and targets (y) for the given data file\n\n    Parameters\n    ----------\n    df_file_path : string\n        File path to generate DataFrame from \n    test : boolean\n        Whether or not the provided data file is the test set\n        False = training set \n        True = test set \n\n    Returns\n    -------\n    pd.DataFrame\n        If it is the test dataset it will return only the features (X)\n        \n    OR \n    \n    Tuple(pd.DataFrame, pd.DataFrame)\n        If it is the training set it will return the features and targets in a tuple (X, y)\n    \"\"\"\n    \n    # read in data and set index to customer ID\n    df = pd.read_feather(df_file_path)\n    df = df.set_index('customer_ID')\n    \n    # get X and y; drop dates from X \n    X = df.drop('S_2', axis=1) if test else df.drop(['S_2', 'target'], axis=1)\n    y = None if test else df['target']\n    \n    # delete original dataframe from memory \n    del df\n    gc.collect()\n    \n    # encode categorical features\n    cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n    encoded_df = one_hot_encode_categorical(X[cat_features], cat_features)\n    \n    # simple impute numerical columns with mean()\n    X = X.drop(cat_features, axis=1)\n    X = simple_impute_numerical(X, list(X.columns))\n    \n    # get final encoded and imputed features\n    X = pd.concat([X, encoded_df], axis=1)\n\n    if test: \n        return X\n    else: \n        return (X, y)\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-02-05T04:20:45.817856Z","iopub.execute_input":"2023-02-05T04:20:45.818489Z","iopub.status.idle":"2023-02-05T04:20:45.829719Z","shell.execute_reply.started":"2023-02-05T04:20:45.818441Z","shell.execute_reply":"2023-02-05T04:20:45.828451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate the features and targets, and then save to .ftr file","metadata":{}},{"cell_type":"code","source":"X_train, y_train = generate_x_y('../input/amexfeather/train_data.ftr')\n\n# sort columns for matching with test set \nX_train = X_train.reindex(sorted(X_train.columns), axis=1)\n\n# feather files do not support indexing\nX_train = X_train.reset_index()\ny_train = y_train.reset_index()\nX_train.to_feather('X_train.ftr')\ny_train.to_feather('y_train.ftr')\n\ndel X_train, y_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T04:20:45.833896Z","iopub.execute_input":"2023-02-05T04:20:45.834710Z","iopub.status.idle":"2023-02-05T04:25:38.614247Z","shell.execute_reply.started":"2023-02-05T04:20:45.834670Z","shell.execute_reply":"2023-02-05T04:25:38.613012Z"},"trusted":true},"execution_count":null,"outputs":[]}]}