{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# pip install xgboost\n# pip install sklearn","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nimport xgboost as xgb # XGBoost typically uses the alias \"xgb\"\nimport numpy as np\nimport sklearn\nimport matplotlib\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import f1_score","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Import train dataset\ndataset = pd.read_csv('train.csv')\n\ndataset.describe()\ndataset.info()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dsToNumeric = dataset.apply(pd.to_numeric, errors='coerce',downcast='integer')\ndsToNumeric.describe()                            ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dsToNumeric['class'].value_counts().plot(kind='bar')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split dataset in train and test\nX, y = dsToNumeric.iloc[:,1:66], dsToNumeric.iloc[:,66]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=.3, random_state=123) \n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Parameter tuning \nscale_pos_weight_opt = np.sum(dsToNumeric['class'] ==0)/np.sum(dsToNumeric['class'] ==1)\n\nparamGrid = {\n    'max_depth': [5,6,7], #list((range(3,12)))\n    'alpha': [ 0.01, .1, 1], #[0, 0.001, 0.01, .1]\n    'subsample': [0.7,1],  #[ 0.25,0.5,0.75, 1]\n    'learning_rate': [0.01,0.1], #np.linspace(0.01 ,0.5,10)\n    'n_estimators': [100,250], #[10,50,100,150,200,250,300]\n    'colsample_bytree': [0.7,0.9], #[0.3,0.5,0.7,0.9,1]\n    'colsample_bylevel': [0.7,0.9], #[0.3,0.5,0.7,0.9,1]\n    #'gamma':[ 0.001, 0.1], is not improving the accuracy and f1 score\n    #'reg_lambda': [ 0.01, 0.1], is not improving the accuracy and f1 score\n    'scale_pos_weight': [scale_pos_weight_opt, 1],\n    'objective': ['binary:logistic'],\n    'eval_metric': ['logloss']\n}\n\nxgb_clf = xgb.XGBClassifier(tree_method='gpu_hist', gpu_id=0)\n# xgb_clf = xgb.XGBClassifier()\n\n# Inspect the parameters\n\n# xgb_clf.get_params\n\ncv = StratifiedKFold()\n\nxgb_rs = GridSearchCV(estimator = xgb_clf, param_grid = paramGrid, cv=cv, verbose=1, scoring='f1')\n\nxgb_rs.fit(X_train, y_train)\n\n#Train the model with the whole dataset for the last kaggle submission\n# xgb_rs.fit(X, y)\n\nprint(\"Best params:\", xgb_rs.best_params_)\nprint(\"Best accuracy:\", xgb_rs.best_score_)\n\n","metadata":{"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Evaluate the model\n\nbest = xgb_rs.best_estimator_\n\n# Plot feature importance\nmatplotlib.rcParams['figure.figsize'] = (10.0, 8)\nxgb.plot_importance(best)\n\n#Predict\npreds = xgb_rs.predict(X_test)\n\nprint(\"Accuracy: \",accuracy_score(y_test, preds))\nprint(\"F1: \", f1_score(y_test, preds))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Change threshold\npreds2 = (xgb_rs.predict_proba(X_test)[:,1] >= 0.3).astype(int)\n\nprint(\"Accuracy: \", accuracy_score(y_test, preds2))\nprint(\"F1: \", f1_score(y_test, preds2))\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Final prediction\n\n#import test.csv\ndataset = pd.read_csv('test.csv')\n#transform to numeric\ndataset = dataset.apply(pd.to_numeric, errors='coerce',downcast='integer')\n\n#fit model\npreds = xgb_rs.predict(dataset.iloc[:,1:66])\n# preds2 = (xgb_rs.predict_proba(dataset.iloc[:,1:66])[:,1] >= 0.3).astype(int)\n\ndataset['class'] = preds\n# dataset['class'] = preds2\n\nfinal = dataset[['id', 'class']]\nfinal.to_csv('sub.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}