{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install xgboost\n\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.datasets import load_iris\nimport xgboost as xgb\nfrom sklearn.metrics import accuracy_score,roc_auc_score, f1_score\n \nfrom sklearn.model_selection import train_test_split, GroupKFold, StratifiedKFold, KFold,cross_val_score, GridSearchCV","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train= pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest= pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\nsub   = pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')\ntrain.head()\n\ntrain.target.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['sex'] = train['sex'].fillna('na')\ntrain['age_approx'] = train['age_approx'].fillna(0)\ntrain['anatom_site_general_challenge'] = train['anatom_site_general_challenge'].fillna('na')\n\ntest['sex'] = test['sex'].fillna('na')\ntest['age_approx'] = test['age_approx'].fillna(0)\ntest['anatom_site_general_challenge'] = test['anatom_site_general_challenge'].fillna('na')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['sex'] = train['sex'].astype(\"category\").cat.codes +1\ntrain['anatom_site_general_challenge'] = train['anatom_site_general_challenge'].astype(\"category\").cat.codes +1\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['sex'] = test['sex'].astype(\"category\").cat.codes +1\ntest['anatom_site_general_challenge'] = test['anatom_site_general_challenge'].astype(\"category\").cat.codes +1\ntest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['patient_id'].shape, train['patient_id'].nunique(), ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['patient_id'].shape, test['patient_id'].nunique(), ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nX = train[['sex', 'age_approx','anatom_site_general_challenge']]\ny = train['target']\n\n\ntest_X = test[['sex', 'age_approx','anatom_site_general_challenge']]\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def runXGB(train_X, train_y, test_X, test_y=None, test_X2=None, seed_val=0, rounds=500, dep=8, eta=0.05):\n    params = {}\n    params[\"objective\"] = \"binary:logistic\"\n    params['eval_metric'] = 'auc'\n    params[\"eta\"] = 0.09\n    params[\"subsample\"] = 0.9\n    params[\"min_child_weight\"] = 1\n    params[\"colsample_bytree\"] = 0.9\n    params[\"max_depth\"] = 4\n    params[\"silent\"] = 1\n    params[\"seed\"] = seed_val\n    params[\"n_estimators\"] = 500\n    params[\"reg_alpha\"] = 0.05\n\n    params[\"gamma\"] = 1\n    num_rounds = rounds\n\n    plst = list(params.items())\n    xgtrain = xgb.DMatrix(train_X, label=train_y)\n\n    xgtest = xgb.DMatrix(test_X, label=test_y)\n    watchlist = [ (xgtrain,'train'), (xgtest, 'test') ]\n    model = xgb.train(plst, xgtrain, num_rounds, watchlist, early_stopping_rounds=200, verbose_eval=500)\n\n\n    pred_test_y = model.predict(xgtest, ntree_limit=model.best_ntree_limit)\n    pred_test_y2 = model.predict(xgb.DMatrix(test_X2), ntree_limit=model.best_ntree_limit)\n    \n    loss = roc_auc_score(test_y, pred_test_y)\n    return pred_test_y, loss, pred_test_y2, model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cv_scores = []\npred_test_full = 0\n\n\n# kf = GroupKFold(n_splits=5)\n# kf = StratifiedKFold(n_splits=5, shuffle=True)\n# kf = StratifiedKFold(n_splits=10, shuffle=True, random_state=30)\nkf = KFold(n_splits=10, shuffle=True, random_state=30)\n\nfor dev_index, val_index in kf.split(X,y):\n    dev_X, val_X = X.loc[dev_index,:], X.loc[val_index,:]\n    dev_y, val_y = y[dev_index], y[val_index]\n\n    \n    pred_val, loss, pred_test, model = runXGB(dev_X, dev_y, val_X, val_y, test_X)\n    \n    f1_scores.append((f1_score(val_y, np.where(pred_val >=0.50,1,0), average='binary')))\n    pred_test_full +=pred_test\n    \n    cv_scores.append(loss)\n    print(cv_scores)\n\npred_test_full /=10.\n\nprint('Avg AUC Score :',sum(cv_scores)/10)\n  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Avg AUC Score : 0.697888987983679","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.target = pred_test_full\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}