{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AMEX-GDBT from Simple to Complex...\n...\n...","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T22:15:17.245283Z","iopub.execute_input":"2022-08-13T22:15:17.245808Z","iopub.status.idle":"2022-08-13T22:15:17.284117Z","shell.execute_reply.started":"2022-08-13T22:15:17.245701Z","shell.execute_reply":"2022-08-13T22:15:17.282777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport gc # Import the garbage collector...\nimport random\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, MinMaxScaler","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:07:57.429141Z","iopub.execute_input":"2022-08-13T23:07:57.429639Z","iopub.status.idle":"2022-08-13T23:07:57.436010Z","shell.execute_reply.started":"2022-08-13T23:07:57.429596Z","shell.execute_reply":"2022-08-13T23:07:57.435105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Configuring the Notebook...","metadata":{}},{"cell_type":"code","source":"%%time\n# Notebook paramters and configuration... \nSEED = 68\nN_FOLDS = 5\nTARGET = ''","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:04:32.581559Z","iopub.execute_input":"2022-08-13T23:04:32.582161Z","iopub.status.idle":"2022-08-13T23:04:32.591208Z","shell.execute_reply.started":"2022-08-13T23:04:32.582101Z","shell.execute_reply":"2022-08-13T23:04:32.589709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# I like to disable my Notebook Warnings To Reduce Noice...\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:17.310646Z","iopub.execute_input":"2022-08-13T22:15:17.311510Z","iopub.status.idle":"2022-08-13T22:15:17.319735Z","shell.execute_reply.started":"2022-08-13T22:15:17.311461Z","shell.execute_reply":"2022-08-13T22:15:17.318302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Notebook Configuration...\n\n# Amount of data we want to load into the Model...\nDATA_ROWS = None\n# Dataframe, the amount of rows and cols to visualize...\nNROWS = 100\nNCOLS = 15\n\n# Main data location path...\nBASE_PATH = '...'","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:17.330481Z","iopub.execute_input":"2022-08-13T22:15:17.331283Z","iopub.status.idle":"2022-08-13T22:15:17.339736Z","shell.execute_reply.started":"2022-08-13T22:15:17.331230Z","shell.execute_reply":"2022-08-13T22:15:17.338326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Seed Python for easy replicability\ndef seed_everything(notebook_seed):\n    random.seed(notebook_seed)\n    np.random.seed(notebook_seed)\n    os.environ['PYTHONHASHSEED'] = str(notebook_seed)\n    \nseed_everything(SEED)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:17.342007Z","iopub.execute_input":"2022-08-13T22:15:17.343154Z","iopub.status.idle":"2022-08-13T22:15:17.352189Z","shell.execute_reply.started":"2022-08-13T22:15:17.343101Z","shell.execute_reply":"2022-08-13T22:15:17.350675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Configure notebook display settings to only use 2 decimal places, tables look nicer...\n\npd.options.display.float_format = '{:,.2f}'.format\npd.set_option('display.max_columns', NCOLS) \npd.set_option('display.max_rows', NROWS)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:17.353972Z","iopub.execute_input":"2022-08-13T22:15:17.354643Z","iopub.status.idle":"2022-08-13T22:15:17.364438Z","shell.execute_reply.started":"2022-08-13T22:15:17.354595Z","shell.execute_reply":"2022-08-13T22:15:17.363344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the Data...","metadata":{}},{"cell_type":"code","source":"%%time\n# Read the Datasets...\n\ntrn_path = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ndef read_dataset(path = ''):\n    '''\n    \n    '''\n    \n    df = pd.read_parquet(path)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:17.379211Z","iopub.execute_input":"2022-08-13T22:15:17.380552Z","iopub.status.idle":"2022-08-13T22:15:17.389443Z","shell.execute_reply.started":"2022-08-13T22:15:17.380488Z","shell.execute_reply":"2022-08-13T22:15:17.388018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Read the Datasets...\n\ntrn_data = read_dataset(trn_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:17.391920Z","iopub.execute_input":"2022-08-13T22:15:17.392867Z","iopub.status.idle":"2022-08-13T22:15:36.333827Z","shell.execute_reply.started":"2022-08-13T22:15:17.392819Z","shell.execute_reply":"2022-08-13T22:15:36.332860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\ntrn_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:36.335705Z","iopub.execute_input":"2022-08-13T22:15:36.337029Z","iopub.status.idle":"2022-08-13T22:15:36.370175Z","shell.execute_reply.started":"2022-08-13T22:15:36.336974Z","shell.execute_reply":"2022-08-13T22:15:36.369222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\ntrn_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:36.371491Z","iopub.execute_input":"2022-08-13T22:15:36.372038Z","iopub.status.idle":"2022-08-13T22:15:36.397997Z","shell.execute_reply.started":"2022-08-13T22:15:36.372001Z","shell.execute_reply":"2022-08-13T22:15:36.396782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\ntrn_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:36.401097Z","iopub.execute_input":"2022-08-13T22:15:36.401481Z","iopub.status.idle":"2022-08-13T22:15:36.411089Z","shell.execute_reply.started":"2022-08-13T22:15:36.401447Z","shell.execute_reply":"2022-08-13T22:15:36.409764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\ntrn_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:15:36.413136Z","iopub.execute_input":"2022-08-13T22:15:36.414020Z","iopub.status.idle":"2022-08-13T22:16:09.929343Z","shell.execute_reply.started":"2022-08-13T22:15:36.413971Z","shell.execute_reply":"2022-08-13T22:16:09.928235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\ntrn_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:16:09.931091Z","iopub.execute_input":"2022-08-13T22:16:09.931454Z","iopub.status.idle":"2022-08-13T22:16:12.129330Z","shell.execute_reply.started":"2022-08-13T22:16:09.931411Z","shell.execute_reply":"2022-08-13T22:16:12.128517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\nfeatures = trn_data.columns\ncat_features = ['B_30','B_38','D_114','D_116','D_117','D_120','D_126','D_63','D_64','D_66','D_68']\nnum_features = [col for col in features if col not in cat_features]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:16:12.130605Z","iopub.execute_input":"2022-08-13T22:16:12.130922Z","iopub.status.idle":"2022-08-13T22:16:12.138044Z","shell.execute_reply.started":"2022-08-13T22:16:12.130894Z","shell.execute_reply":"2022-08-13T22:16:12.136643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\ndef aggregate_features(df, selected_cols, agg_field = 'customer_ID', types = ['first']):\n    '''\n    This function creates...\n    Args:\n        ...\n        ...\n        ...\n        ...\n        ...\n    \n    Returns:\n        ...\n    '''\n    \n    # ...\n    agg_dataset = df.groupby(agg_field)[selected_cols].agg(types)\n    \n    # ...\n    agg_dataset.columns = ['_'.join(x) for x in agg_dataset.columns]\n    agg_dataset.reset_index(inplace = True)\n    \n    return agg_dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:16:12.139926Z","iopub.execute_input":"2022-08-13T22:16:12.140320Z","iopub.status.idle":"2022-08-13T22:16:12.152962Z","shell.execute_reply.started":"2022-08-13T22:16:12.140284Z","shell.execute_reply":"2022-08-13T22:16:12.151884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\ntrn_data_numagg = aggregate_features(trn_data, selected_cols = num_features, agg_field = 'customer_ID', types = ['first', 'mean', 'std', 'min', 'max', 'last'])\ntrn_data_catagg = aggregate_features(trn_data, selected_cols = cat_features, agg_field = 'customer_ID', types = ['count', 'first', 'last', 'nunique'])\n\ntrn_data_agg = trn_data_numagg.merge(trn_data_catagg, how = 'left', on = 'customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:16:12.155071Z","iopub.execute_input":"2022-08-13T22:16:12.155585Z","iopub.status.idle":"2022-08-13T22:21:38.599729Z","shell.execute_reply.started":"2022-08-13T22:16:12.155532Z","shell.execute_reply":"2022-08-13T22:21:38.598154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Release some of the memory used...\n\ndel trn_data, trn_data_numagg, trn_data_catagg\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:21:38.603622Z","iopub.execute_input":"2022-08-13T22:21:38.604901Z","iopub.status.idle":"2022-08-13T22:21:38.812152Z","shell.execute_reply.started":"2022-08-13T22:21:38.604840Z","shell.execute_reply":"2022-08-13T22:21:38.810691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Some Comments...\n\ndef encode_variables(train_df, test_df, cat_columns):\n    '''\n    This function creates...\n    Args:\n        ...\n        ...\n        ...\n        ...\n        ...\n    \n    Returns:\n        ...\n    '''\n    encoder = LabelEncoder()\n    for col in df[cat_columns].columns:\n        train_df[cat_col] = encoder.fit_transform(train_df[cat_col])\n        test_df[cat_col] = encoder.fit_transform(test_df[cat_col])\n    \n    return train_df, test_df","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:21:38.816247Z","iopub.execute_input":"2022-08-13T22:21:38.816896Z","iopub.status.idle":"2022-08-13T22:21:38.829093Z","shell.execute_reply.started":"2022-08-13T22:21:38.816856Z","shell.execute_reply":"2022-08-13T22:21:38.827351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_ml_model(train_df, test_df, features):\n    '''\n    \n    '''\n    skf = StratifiedKFold(n_splits = N_FOLDS, shuffle = True, random_state = SEED)\n    \n    for fold, (trn_idx, val_idx) in enumerate(skf.split(train_df, train_df[TARGET])):\n        \n        print(f'Training Fold: {fold} ...')\n        \n        # ...\n        X_trn, X_val = train_df[features].iloc[trn_idx], train_df[features].iloc[val_idx]  \n        y_trn, y_val = train_df[TARGET].iloc[trn_idx], train_df[TARGET].iloc[val_idx]  \n        \n        # ...\n        scaler = StandardScaler()\n        X_trn_scaled = scaler.fit_transform(X_trn)\n        X_val_scaled = scaler.transform(X_val)\n        \n        # ...\n        \n        \n        # ...\n        \n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:21:38.830576Z","iopub.execute_input":"2022-08-13T22:21:38.831133Z","iopub.status.idle":"2022-08-13T22:21:38.838676Z","shell.execute_reply.started":"2022-08-13T22:21:38.831098Z","shell.execute_reply":"2022-08-13T22:21:38.837518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tst_data = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T22:21:38.840194Z","iopub.execute_input":"2022-08-13T22:21:38.840856Z","iopub.status.idle":"2022-08-13T22:21:38.851383Z","shell.execute_reply.started":"2022-08-13T22:21:38.840819Z","shell.execute_reply":"2022-08-13T22:21:38.850228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}