{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Refer this notebook for modeling part: https://www.kaggle.com/awaldeep/xgboost-optuna-baseline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-25T19:12:32.673012Z","iopub.execute_input":"2022-06-25T19:12:32.673488Z","iopub.status.idle":"2022-06-25T19:12:32.705092Z","shell.execute_reply.started":"2022-06-25T19:12:32.673392Z","shell.execute_reply":"2022-06-25T19:12:32.704067Z"}}},{"cell_type":"code","source":"# Import libraries\nimport os\nimport warnings\n\nimport numpy as np\nimport pandas as pd\n\nimport gc  # Garbage collector\n\n\nwarnings.filterwarnings('ignore')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GPU libraries\nimport cupy, cudf ","metadata":{"execution":{"iopub.status.busy":"2022-06-25T19:13:10.359885Z","iopub.execute_input":"2022-06-25T19:13:10.360213Z","iopub.status.idle":"2022-06-25T19:13:14.704172Z","shell.execute_reply.started":"2022-06-25T19:13:10.360137Z","shell.execute_reply":"2022-06-25T19:13:14.703008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom xgboost import XGBClassifier","metadata":{"execution":{"iopub.status.busy":"2022-06-25T19:12:36.298537Z","iopub.execute_input":"2022-06-25T19:12:36.298981Z","iopub.status.idle":"2022-06-25T19:12:37.002007Z","shell.execute_reply.started":"2022-06-25T19:12:36.298934Z","shell.execute_reply":"2022-06-25T19:12:37.000687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the model\nimport joblib\n# xgb_classifier = joblib.load(\"../input/01-starter-xgboost-implementation/xgb_classifier_v1.h5\")\nfinal_model = joblib.load(\"../input/xgboost-optuna-baseline/xgb_classifier_v1.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-06-25T19:13:14.708917Z","iopub.execute_input":"2022-06-25T19:13:14.710086Z","iopub.status.idle":"2022-06-25T19:13:14.894602Z","shell.execute_reply.started":"2022-06-25T19:13:14.710043Z","shell.execute_reply":"2022-06-25T19:13:14.89376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.get_xgb_params()","metadata":{"execution":{"iopub.status.busy":"2022-06-25T19:13:17.618969Z","iopub.execute_input":"2022-06-25T19:13:17.61984Z","iopub.status.idle":"2022-06-25T19:13:17.63285Z","shell.execute_reply.started":"2022-06-25T19:13:17.619803Z","shell.execute_reply":"2022-06-25T19:13:17.631788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-25T19:14:16.353657Z","iopub.execute_input":"2022-06-25T19:14:16.35456Z","iopub.status.idle":"2022-06-25T19:14:16.364917Z","shell.execute_reply.started":"2022-06-25T19:14:16.35452Z","shell.execute_reply":"2022-06-25T19:14:16.363986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_test_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    #df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(0) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading test data...')\nTEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\ntest = read_test_file(path = TEST_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T19:14:21.210463Z","iopub.execute_input":"2022-06-25T19:14:21.210834Z","iopub.status.idle":"2022-06-25T19:15:03.272282Z","shell.execute_reply.started":"2022-06-25T19:14:21.210803Z","shell.execute_reply":"2022-06-25T19:15:03.271255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = process_and_feature_engineer(test)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T19:15:03.274437Z","iopub.execute_input":"2022-06-25T19:15:03.274982Z","iopub.status.idle":"2022-06-25T19:15:11.419326Z","shell.execute_reply.started":"2022-06-25T19:15:03.274924Z","shell.execute_reply":"2022-06-25T19:15:11.41829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['prediction'] = final_model.predict_proba(test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-06-25T11:19:10.365832Z","iopub.execute_input":"2022-06-25T11:19:10.366285Z","iopub.status.idle":"2022-06-25T11:19:13.850013Z","shell.execute_reply.started":"2022-06-25T11:19:10.366249Z","shell.execute_reply":"2022-06-25T11:19:13.848561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = pd.DataFrame(test['prediction'].to_pandas())","metadata":{"execution":{"iopub.status.busy":"2022-06-25T11:19:16.431577Z","iopub.execute_input":"2022-06-25T11:19:16.432174Z","iopub.status.idle":"2022-06-25T11:19:17.033524Z","shell.execute_reply.started":"2022-06-25T11:19:16.432122Z","shell.execute_reply":"2022-06-25T11:19:17.031988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final.to_csv(\"submission.csv\", index=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T11:19:17.041802Z","iopub.execute_input":"2022-06-25T11:19:17.04594Z","iopub.status.idle":"2022-06-25T11:19:21.565013Z","shell.execute_reply.started":"2022-06-25T11:19:17.045853Z","shell.execute_reply":"2022-06-25T11:19:21.563622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}