{"cells":[{"metadata":{},"cell_type":"markdown","source":"Importing All useful Libraries","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_meta = pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\")\ntest_meta = pd.read_csv(\"../input/siim-isic-melanoma-classification/test.csv\")\n\nuseful_cols = ['sex',\"age_approx\",\"anatom_site_general_challenge\"]\nTARGET = \"target\"\nID = \"image_name\"\ntrain_meta = train_meta[useful_cols+[TARGET,ID]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_meta.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We see that we have some Nan values in repective columns , so as test data also contains anatom_site_general_challenge wit some nan values ,hence we fill a new class into these nan positions and drop records where age or sex are provided as nan.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_meta['anatom_site_general_challenge'] = train_meta['anatom_site_general_challenge'].fillna(\"unknown_site\")\ntest_meta['anatom_site_general_challenge'] = test_meta['anatom_site_general_challenge'].fillna(\"unknown_site\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_meta.dropna(inplace=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We use LabelEncoder to encode the string columns present in our dataset. and store respective encoders for each column into a dictionary so that it can be used in test data also.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom collections import defaultdict\n\nencoders = defaultdict(LabelEncoder)\nfor column in train_meta.select_dtypes(\"object\").columns:\n    if column in [ID,TARGET]:\n        continue\n    encoder = LabelEncoder()\n    train_meta[column] = encoder.fit_transform(train_meta[[column]])\n    encoders[column] = encoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_meta.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Now comes the main model part.\nI have implemented 5 kfold ,to examine my model performance.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"X = train_meta.drop([ID,TARGET],axis=1)\nY = train_meta[[TARGET]]\n\nfrom sklearn.model_selection import KFold\nfolds = KFold(n_splits=5,shuffle=True)\n\nparams ={\n    \"od_type\":\"Iter\",\n    'od_wait':100,\n    \"eval_metric\":\"AUC\",\n    'loss_function':'Logloss',\n    \"iterations\":1000,\n    \"verbose\":100\n}\n\nscores = []\n\nmax_score = -np.inf\nfor (train_idx,test_idx),i in zip(folds.split(X,Y),range(0,5)):\n    print(\"Working On fold \",i)\n    model = CatBoostClassifier(**params)\n    model.fit(X.iloc[train_idx],Y.iloc[train_idx],\n              eval_set=(X.iloc[test_idx],Y.iloc[test_idx]),\n              cat_features = [\"sex\",\"anatom_site_general_challenge\"])\n    \n    score = model.score(model.predict(X.iloc[test_idx]),Y.iloc[test_idx])\n    print(\"Achieved AUC Score :\" ,score)\n    scores.append(score)\n    print(scores)\n    if score > max_score:\n        best_idx = (train_idx,test_idx)\n        max_score = score\n        \n    print(\"-\"*100)\n    \n\nprint(\"Final Results from 5 KFOLD\")\nprint(\"Min Score\",min(scores))\nprint(\"Mean Score\",sum(scores)/len(scores))\nprint(\"Max Score\",max(scores))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"I have used the best indices provided from above experiment to prepare my model for final submission","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"model = CatBoostClassifier(**params)\nmodel.fit(X.iloc[best_idx[0]],Y.iloc[best_idx[0]],\n              eval_set=(X.iloc[best_idx[1]],Y.iloc[best_idx[1]]),\n              cat_features = [\"sex\",\"anatom_site_general_challenge\"])\n\nscore = model.score(model.predict(X.iloc[test_idx]),Y.iloc[test_idx])\nprint(\"Achieved AUC Score :\" ,score)\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Submission Time","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"converting classes back to there respective encoded values using LabelEncoders we trained before.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"for key in encoders.keys():\n    test_meta[key] = encoders[key].transform(test_meta[key])\ntest_meta.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Final Touch to submit  csv file.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"testing = test_meta.drop([ID,'patient_id'],axis=1)\npredictions = model.predict_proba(testing)\npredictions = [i[1] for i in predictions]\ntest_meta[TARGET] = predictions\ntest_meta[[ID,TARGET]].to_csv(\"catboost_submission.csv\",index=None)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"With this code , i was able to achieve 0.7 score on public leaderboard.\nI know this is not much to be used , but atleast it can be useful for some one.\n\nWill be updating this notebook after hyperparameter tuning .\n\nKindly comment if you like or dislike something in this notebook, and if you lked please upvote it too.\nand also do share some suggestions which can be used to improve it.\n\nThank You.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}