{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport time\nimport xgboost as xgb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T02:52:19.263856Z","iopub.execute_input":"2022-07-30T02:52:19.265281Z","iopub.status.idle":"2022-07-30T02:52:20.524882Z","shell.execute_reply.started":"2022-07-30T02:52:19.264929Z","shell.execute_reply":"2022-07-30T02:52:20.523989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = pd.read_csv('../input/xgb-fraud-with-magic-0-9600/X_train.csv')\nX_test = pd.read_csv('../input/xgb-fraud-with-magic-0-9600/X_test.csv')\ny_train = pd.read_csv('../input/xgb-fraud-with-magic-0-9600/y_train.csv',header=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:52:20.526614Z","iopub.execute_input":"2022-07-30T02:52:20.527303Z","iopub.status.idle":"2022-07-30T02:53:21.696705Z","shell.execute_reply.started":"2022-07-30T02:52:20.527247Z","shell.execute_reply":"2022-07-30T02:53:21.695781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_auc(y_true, y_prob):\n    y_true = np.asarray(y_true)\n    y_true = y_true[np.argsort(y_prob)]\n    nfalse = 0\n    auc = 0\n    n = len(y_true)\n    for i in range(n):\n        y_i = y_true[i]\n        nfalse += (1 - y_i)\n        auc += y_i * nfalse\n    auc = np.array(auc, dtype='f')\n    auc /= (nfalse * (n - nfalse))\n    return auc","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:53:41.516940Z","iopub.execute_input":"2022-07-30T02:53:41.517286Z","iopub.status.idle":"2022-07-30T02:53:41.523937Z","shell.execute_reply.started":"2022-07-30T02:53:41.517257Z","shell.execute_reply":"2022-07-30T02:53:41.522693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\n\nkf = KFold(n_splits=10)\nfold_score = []\nfor fold, (train_idx, test_idx) in enumerate(kf.split(X_train)):\n    clf = xgb.XGBClassifier( \n        objective='binary:logistic',\n        n_estimators=500,\n        max_depth=12, \n        learning_rate=0.02, \n        subsample=0.8,\n        colsample_bytree=0.4, \n        eval_metric=['auc','logloss'],\n        nthread=4,\n        tree_method='hist' \n    )\n    start_time = time.time()\n    clf.fit(X_train.loc[train_idx], y_train.iloc[:,1][train_idx],\n           eval_set=[(X_train.loc[test_idx],y_train.iloc[:,1][test_idx])],verbose=50)\n    print(time.time()-start_time)\n    fold_score.append(fast_auc(y_train.iloc[:,1][test_idx],clf.predict_proba(X_train.loc[test_idx])[:,1]))\n    clf.save_model('model_fold_' +str(fold)+ '.json')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:54:04.525076Z","iopub.execute_input":"2022-07-30T02:54:04.525460Z","iopub.status.idle":"2022-07-30T03:01:09.162328Z","shell.execute_reply.started":"2022-07-30T02:54:04.525427Z","shell.execute_reply":"2022-07-30T03:01:09.160694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\n\nxgb_params = {\n    'eta': 0.02,\n    'booster' : \"gbtree\",\n    'max_depth': 12,\n    'subsample': 0.8,\n    'colsample_bytree': 0.4,\n    'objective': \"binary:logistic\",\n    'eval_metric': ['auc','logloss'],\n    'nthread':4,\n    'tree_method':'hist'\n}\n\nkf = KFold(n_splits=10)\nfold_score = []\nfor fold, (train_idx, test_idx) in enumerate(kf.split(X_train)):\n    dtrain = xgb.DMatrix(X_train.loc[train_idx], y_train.iloc[:,1][train_idx])\n    dvalid = xgb.DMatrix(X_train.loc[test_idx],y_train.iloc[:,1][test_idx])\n\n    start_time = time.time()\n#     clf.fit(X_train.loc[train_idx], y_train.iloc[:,1][train_idx],\n#            eval_set=[(X_train.loc[test_idx],y_train.iloc[:,1][test_idx])],verbose=50)\n    model = xgb.train(xgb_params, dtrain,evals=[(dvalid,'valid')], num_boost_round=500,verbose_eval=50)\n    print(time.time()-start_time)\n    fold_score.append(fast_auc(y_train.iloc[:,1][test_idx],clf.predict_proba(X_train.loc[test_idx])[:,1]))\n    clf.save_model('model_fold_' +str(fold)+ '.json')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T03:01:43.013796Z","iopub.execute_input":"2022-07-30T03:01:43.014383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print('Cross validation score = %1.5f' % np.mean(fold_score))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result = []\n\n# for fold in range(10):\n\n#     clf = xgb.XGBClassifier()\n#     clf.load_model('model_fold_' +str(fold)+ '.json')\n#     result.append(clf.predict_proba(X_test)[:,1])","metadata":{"execution":{"iopub.status.busy":"2021-10-16T11:30:53.71451Z","iopub.execute_input":"2021-10-16T11:30:53.715051Z","iopub.status.idle":"2021-10-16T11:30:59.682203Z","shell.execute_reply.started":"2021-10-16T11:30:53.714983Z","shell.execute_reply":"2021-10-16T11:30:59.681303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_df = pd.read_csv('../input/ieee-fraud-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-10-16T11:31:01.568869Z","iopub.execute_input":"2021-10-16T11:31:01.569244Z","iopub.status.idle":"2021-10-16T11:31:01.795472Z","shell.execute_reply.started":"2021-10-16T11:31:01.569211Z","shell.execute_reply":"2021-10-16T11:31:01.794336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_df['isFraud'] = np.mean(result,axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-10-16T11:31:03.007887Z","iopub.execute_input":"2021-10-16T11:31:03.008285Z","iopub.status.idle":"2021-10-16T11:31:03.019048Z","shell.execute_reply.started":"2021-10-16T11:31:03.008249Z","shell.execute_reply":"2021-10-16T11:31:03.018155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_df.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}