{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# now we implement a model that will predict based on the metadata given in the original data. \n\n!pip install xgboost\nimport xgboost as xgb\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import accuracy_score\nimport keras\nimport cv2 # computer vision library¨\nimport tensorflow as tf # machine learning library\nimport matplotlib.pyplot as plt # data visualization tool\nfrom tensorflow.python.keras import backend as K #to utilize more of keras' functionality","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\nsub = pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['sex'] = train['sex'].astype(\"category\").cat.codes +1\ntrain['anatom_site_general_challenge'] = train['anatom_site_general_challenge'].astype(\"category\").cat.codes +1\ntest['sex'] = test['sex'].astype(\"category\").cat.codes +1\ntest['anatom_site_general_challenge'] = test['anatom_site_general_challenge'].astype(\"category\").cat.codes +1\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":" train['sex'] = train['sex'].fillna('male')\ntrain['age_approx'] = train['age_approx'].fillna(train['age_approx'].mean())\ntrain['anatom_site_general_challenge'] = train['anatom_site_general_challenge'].fillna('torso')\ntest['sex'] = test['sex'].fillna('male')\ntest['age_approx'] = test['age_approx'].fillna(train['age_approx'].mean())\ntest['anatom_site_general_challenge'] = test['anatom_site_general_challenge'].fillna('torso')\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#to have a more balanced dataset, we create a new dataframe that contains a more equal percentage of each type of target image\ndf_0=train[train['target']==0].sample(600)\ndf_1=train[train['target']==1]\ntrain=pd.concat([df_0,df_1])\ntrain=train.reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train_df = train[['sex', 'age_approx','anatom_site_general_challenge']]\ny_train_df = train['target']\n\n\nx_train, x_val, y_train, y_val = train_test_split(x_train_df, y_train_df, test_size=0.2, random_state=42)\n\n\nx_test = test[['sex', 'age_approx','anatom_site_general_challenge']]\n\n\ntrain_DMatrix = xgb.DMatrix(x_train, label= y_train)\ntest_DMatrix = xgb.DMatrix(x_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clf = xgb.XGBClassifier(n_estimators=2000, \n                        max_depth=10, \n                        objective='multi:softprob',\n                        seed=0,  \n                        #nthread=-1, \n                        learning_rate=0.15, \n                        num_class = 2, \n                        scale_pos_weight = (32542/584))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clf.fit(x_train, y_train)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = clf.predict_proba(x_val)[:,1]\ntarget = [round(value) for value in preds]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# evaluate predictions\naccuracy = accuracy_score(y_val, target)\nprint(\"Accuracy: %.2f%%\" % (accuracy * 100.0))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#the submission file:\ntest_df = test[['sex', 'age_approx','anatom_site_general_challenge']]\npreds = clf.predict_proba(test_df)[:,1]\n\n#target = [round(value) for value in preds]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#fetch data from CNN\ndf = pd.read_csv(\"../input/cnn-model-predictions/submission2.csv\")\ndf =df[\"target\"].values.tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"els = []\nfor i in range(0,len(preds)):\n    els.append(preds[i]*0.2+0.8*df[i])\nsub[\"target\"] = els\nsub.to_csv('submission.csv', index=False)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}