{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"'''\nDatasets are available at:- https://www.kaggle.com/datasets/ydvaakash/new-amex\n'''\nimport pandas as pd\nimport numpy as np\nfrom sklearn import model_selection\nimport gc; gc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-20T19:41:11.273649Z","iopub.execute_input":"2022-09-20T19:41:11.274116Z","iopub.status.idle":"2022-09-20T19:41:12.725854Z","shell.execute_reply.started":"2022-09-20T19:41:11.274076Z","shell.execute_reply":"2022-09-20T19:41:12.724543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traini = pd.read_csv('../input/amex-default-prediction/train_data.csv', parse_dates=['S_2'],chunksize=400_000, iterator=True)\n\ntrain = []\nfor df in traini:\n    if len(train) > 0:\n        train = pd.concat([train, df])\n    else:\n        train = df[:]\n    del df;gc.collect()\n\ntrain.sort_values(by=['S_2'], inplace=True)\ntrain.reset_index(drop=True, inplace=True)\n\n#denoising data according to https://www.kaggle.com/code/raddar/amex-data-int-types-train\nfor col in train.columns:\n    if col not in ['customer_ID','S_2','D_63','D_64']:\n        train[col] = np.floor(train[col]*100)\n\n        \ntrain['D_63'] = train['D_63'].apply(lambda t: {'CR':0, 'XZ':1, 'XM':2, 'CO':3, 'CL':4, 'XL':5}[t]).astype(np.int8)\ntrain['D_64']=train['D_64'].fillna(242)\ntrain['D_64'] = train['D_64'].apply(lambda t: {'O':0, '-1':1, 'R':2, 'U':3, 242:-1}[t]).astype(np.int8)\n\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\ndate_cols = ['S_2']\nstr_cols=['customer_ID','D_63', 'D_64']\n\nfor cf in str_cols:\n    train[cf] = train[cf].astype(str)\n\nfor cols in train.columns:\n    if cols not in str_cols+date_cols+cat_cols:\n        if (train[cols].dtype.name =='float64'):\n            train[cols] = train[cols].astype(np.float16)\ntrain=train.fillna(-1)\n\nlabels=pd.read_csv('../input/amex-default-prediction/train_labels.csv')\ntrain=train.merge(labels,how='left', on='customer_ID')\ntrain[\"kfold\"] = -1\nkf = model_selection.StratifiedKFold(n_splits=5,shuffle=True, random_state=42)\n\nfor fold, (train_indicies, valid_indicies) in enumerate(kf.split(X=train,y=train.target)):\n    train.loc[valid_indicies, \"kfold\"] = fold\n\ntrain.to_feather('amex_train.feather', compression='zstd', compression_level=2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testi = pd.read_csv('../input/amex-default-prediction/test_data.csv', parse_dates=['S_2'],chunksize=400_000, iterator=True)\n\ntest = []\nfor df in testi:\n    if len(test) > 0:\n        test = pd.concat([test, df])\n    else:\n        test = df[:]\n    del df;gc.collect()\n\ntest.sort_values(by=['S_2'], inplace=True)\ntest.reset_index(drop=True, inplace=True)\n\n\n\nfor col in test.columns:\n    if col not in ['customer_ID','S_2','D_63','D_64']:\n        test[col] = np.floor(test[col]*100)\n        \ntest['D_63'] = test['D_63'].apply(lambda t: {'CR':0, 'XZ':1, 'XM':2, 'CO':3, 'CL':4, 'XL':5}[t]).astype(np.int8)\ntest['D_64']=test['D_64'].fillna(242)\ntest['D_64'] = test['D_64'].apply(lambda t: {'O':0, '-1':1, 'R':2, 'U':3, 242:-1}[t]).astype(np.int8)\n\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\ndate_cols = ['S_2']\nstr_cols=['customer_ID','D_63', 'D_64']\n\nfor cf in str_cols:\n    test[cf] = test[cf].astype(str)\n    \nfor cols in test.columns:\n    if cols not in str_cols+date_cols+cat_cols:\n        if (test[cols].dtype.name =='float64'):\n            test[cols] = test[cols].astype(np.float32)\n            \ntest=test.fillna(-1)\n\ntest.to_feather('amex_test.feather', compression='zstd', compression_level=2)","metadata":{},"execution_count":null,"outputs":[]}]}