{"cells":[{"metadata":{},"cell_type":"markdown","source":"### Libraries","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Basic numerical and other libraries\nimport random\nimport numpy as np\nimport pandas as pd\nfrom scipy import stats\nimport sys\nimport functools\n\n# Display option\nfrom IPython.display import display, HTML\npd.options.display.max_rows = 5000\npd.options.display.max_columns = 5000\n\n# Handle warnings (during execution of code)\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Datetime\nimport time\nfrom datetime import datetime\nfrom datetime import timedelta\n\n# Visulisation\nfrom pprint import pprint\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# statsmodel\nimport statsmodels.api as statsm\nimport statsmodels.discrete.discrete_model as sm\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\n\n# Imbalanced Data Handling\nimport imblearn\nfrom imblearn.over_sampling import RandomOverSampler\nfrom imblearn.under_sampling import RandomUnderSampler\n\n# sklearn\nfrom sklearn.model_selection import StratifiedKFold, KFold, LeaveOneOut, cross_val_score, GridSearchCV, RandomizedSearchCV, train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn import tree\nfrom sklearn.metrics import auc, roc_auc_score, roc_curve\n\n\n# xgboost, lightgbm\nfrom xgboost import XGBClassifier","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### User-Defined Functions","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_stats(df, feature, target_ftr, label_rotation=False, horizontal_layout=True):\n    '''\n    This function plot the categorical feature distribution according to target variable\n    '''\n    temp = df[feature].value_counts()\n    df1 = pd.DataFrame({feature: temp.index,'Number of Patients': temp.values})\n\n    # Calculate the percentage of target=1 per category value\n    cat_perc = df[[feature, target_ftr]].groupby([feature],as_index=False).mean()\n    cat_perc.sort_values(by=target_ftr, ascending=False, inplace=True)\n    \n    sns.set_color_codes(\"pastel\")\n    s = sns.barplot(x = feature, y=\"Number of Patients\",data=df1)\n    if(label_rotation):\n        s.set_xticklabels(s.get_xticklabels(),rotation=60)\n\n    plt.tick_params(axis='both', which='major', labelsize=10)\n\n    plt.show();\n\n\ndef get_rocauc(model, xTest, yTest): \n    '''\n    This function produces the Area under the curve for the model. \n    The 'auto' method calculates this metric by using the roc_auc_score function from sklearn.\n    Range: 0 to 1 (0 being the worst predictive model, 0.5 being the random and 1 being the best)\n    '''\n    predictions = model.predict_proba(xTest)[:, 1]\n    roc_auc = roc_auc_score(yTest, predictions)\n    print('Model Performance:')\n    print('--'*5)\n    print('--'*5)\n    print('ROC = {:0.2f}%'.format(roc_auc))\n    \n    return roc_auc","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Import Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ndf_test  = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\ndf_sub   = pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### 1. #records and #features","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# shape of training, test and submission data\n\nprint(f'training data shape: {df_train.shape}')\nprint(f'test data shape: {df_test.shape}')\nprint(f'submission data shape: {df_sub.shape}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Need to analyse how more than 1 feature is less in test data than in training data.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### How does each of the datasets look like ?","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### train dataset","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### test dataset","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"df_test.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* 'diagnosis' and 'benign_malignant' are missing in test dataset. \n* So while building any algorithm we will not be able to use these 2 features.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### submission dataset","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"df_sub.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Here, we may need to replace target with the predicted values from the trained model.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Let's check the unique values in the targets of training and submission datasets","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# training dataset\n\ndf_train['target'].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# submission dataset\n\ndf_sub['target'].unique()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* target values in training dataset looks good and in submission each and every image is assumed as benign.\n* Our job is to predict the probability of an image being malignant.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### EDA","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Target Class Distribution","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"sns.countplot(df_train.target)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"df_train['target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Huge Class imbalance is there.\n* We may use some sampling techniques to treat the class-imbalance. (**Will try after running the baseline model)","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Gender Distribution","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# Gender Distribution\n\nplot_stats(df_train, 'sex', 'target', label_rotation=False, horizontal_layout=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Train data consists a very balanced gender distribution of patients.\n","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### anatom_site_general_challenge Distribution","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# anatom_site_general_challenge Distribution\n\nplot_stats(df_train, 'anatom_site_general_challenge', 'target', label_rotation=90, horizontal_layout=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Age Distribution of Patients according to target","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# Age Distribution of Patients according to target\n\nfig, axes = plt.subplots(1, 2)\n\nfig.set_size_inches(12, 4)\n\ndf_train[df_train['target']==0].hist('age_approx', bins=100, ax=axes[0])\naxes[0].set_xlabel('benign')\ndf_train[df_train['target']==1].hist('age_approx', bins=100, ax=axes[1])\naxes[1].set_xlabel('malignant')\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* This can help in distinguishing between benign and malignant as median is less for benign","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### Data Preprocessing","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head(3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Let's check the null values","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Missing Values","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"* Training Data","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# Calculate missing value count and percentage\n\nmissing_value_df_train = pd.DataFrame(index = df_train.keys(), data =df_train.isnull().sum(), columns = ['Missing_Value_Count'])\nmissing_value_df_train['Missing_Value_Percentage'] = ((df_train.isnull().mean())*100)\nmissing_value_df_train.sort_values('Missing_Value_Count',ascending= False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Test Data","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# Calculate missing value count and percentage\n\nmissing_value_df_test = pd.DataFrame(index = df_test.keys(), data =df_test.isnull().sum(), columns = ['Missing_Value_Count'])\nmissing_value_df_test['Missing_Value_Percentage'] = ((df_test.isnull().mean())*100)\nmissing_value_df_test.sort_values('Missing_Value_Count',ascending= False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* we need to impute these 3 features    \n\n  1. anatom_site_general_challenge\n  2. age_approx\n  3. sex","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Imputation","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"* Training Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Replace age with median\n\nage_array = df_train[df_train[\"age_approx\"]!=np.nan][\"age_approx\"]\ndf_train[\"age_approx\"].replace(np.nan, age_array.median(), inplace=True)\n\n\n# Replace sex and anatom_site_general_challenge with mode\n\nsex_array = df_train[~(df_train[\"sex\"].isnull())][\"sex\"]\ndf_train['sex'].fillna(sex_array.mode().values[0], inplace=True)\n\nanatom_array = df_train[~(df_train[\"anatom_site_general_challenge\"].isnull())][\"anatom_site_general_challenge\"]\ndf_train['anatom_site_general_challenge'].fillna(anatom_array.mode().values[0], inplace=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Test Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Replace age with median for test data\n\n# age_array = df_test[df_test[\"age_approx\"]!=np.nan][\"age_approx\"]\n# df_test[\"age_approx\"].replace(np.nan, age_array.median(), inplace=True)\n\n\n# # Replace sex and anatom_site_general_challenge with mode\n\n# sex_array = df_test[~(df_test[\"sex\"].isnull())][\"sex\"]\n# df_test['sex'].fillna(sex_array.mode().values[0], inplace=True)\n\nanatom_array_test = df_test[~(df_test[\"anatom_site_general_challenge\"].isnull())][\"anatom_site_general_challenge\"]\ndf_test['anatom_site_general_challenge'].fillna(anatom_array_test.mode().values[0], inplace=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Categorical Feature Handling","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Unique values for feature 'sex'\n\nprint('Unique values for gender:')\nprint(df_train['sex'].unique())\nprint('--'*20)\nprint('--'*20)\nprint('Unique values for anatom_site_general_challenge:')\nprint(df_train['anatom_site_general_challenge'].unique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Training Data ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"## Feature 'sex'\n\n# Need to convert the datatypes of the feature to 'category' before Label encoding.\ndf_train[\"sex\"] = df_train[\"sex\"].astype('category')\n\n# Label Encoding\ndf_train[\"sex_cat\"] = df_train[\"sex\"].cat.codes\n\ndf_train[[\"sex\", \"sex_cat\"]].head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Feature 'anatom_site_general_challenge'\n\n# Need to convert the datatypes of the feature to 'category' before Label encoding.\ndf_train[\"anatom_site_general_challenge\"] = df_train[\"anatom_site_general_challenge\"].astype('category')\n\n# Label Encoding\ndf_train[\"anatom_site_general_challenge_cat\"] = df_train[\"anatom_site_general_challenge\"].cat.codes\n\ndf_train[[\"anatom_site_general_challenge\", \"anatom_site_general_challenge_cat\"]].head(3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Test Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"### Categorical Data Handling for test data\n\n\n## Feature 'sex'\n\n# Need to convert the datatypes of the feature to 'category' before Label encoding.\ndf_test[\"sex\"] = df_test[\"sex\"].astype('category')\n\n# Label Encoding\ndf_test[\"sex_cat\"] = df_test[\"sex\"].cat.codes\n\ndf_test[[\"sex\", \"sex_cat\"]].head(3)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### Categorical Data Handling for test data\n\n## Feature 'anatom_site_general_challenge'\n\n# Need to convert the datatypes of the feature to 'category' before Label encoding.\ndf_test[\"anatom_site_general_challenge\"] = df_test[\"anatom_site_general_challenge\"].astype('category')\n\n# Label Encoding\ndf_test[\"anatom_site_general_challenge_cat\"] = df_test[\"anatom_site_general_challenge\"].cat.codes\n\ndf_test[[\"anatom_site_general_challenge\", \"anatom_site_general_challenge_cat\"]].head(3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Feature Selection","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# feature set\n\nftr_set = ['sex_cat',\n           'age_approx',\n           'anatom_site_general_challenge_cat']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Training, Validation & Test Data Preparation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# dependent and independent features of training and test datasets\n\nexog_train = df_train[ftr_set]\nendog_train = df_train['target']\n\nexog_test = df_test[ftr_set]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* We need to prepare a validation set for simple baseline accuracy calculation.\n* I will try to make as small as possible the validation set as 33k records are there in training dataset.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### Imbalanced Data Handling","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Oversampling","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# define oversampling strategy\noversample = RandomOverSampler(sampling_strategy=0.7)\nX_over, y_over = oversample.fit_resample(exog_train, endog_train)\n\nX_over = pd.DataFrame(X_over)\ny_over = pd.DataFrame(y_over)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"# print(X_over.shape)\n# print(y_over.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Undersampling","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# define undersampling strategy\nundersample = RandomUnderSampler(sampling_strategy=0.1)\nX_under, y_under = undersample.fit_resample(exog_train, endog_train)\n\nX_under = pd.DataFrame(X_under)\ny_under = pd.DataFrame(y_under)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"# print(X_under.shape)\n# print(y_under.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Training & Validation data split","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"* Oversampled","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Oversampled train and validation split\n\nx_train, x_val, y_train, y_val = train_test_split(X_over, y_over, test_size=0.2, stratify=y_over, random_state=42)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Undersampled","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Undersampled train and validation split\n\nx_train, x_val, y_train, y_val = train_test_split(X_under, y_under, test_size=0.2, stratify=y_under, random_state=42)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Insample (Without imbalanced class treatment)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Insample train and validation split\n\nx_train, x_val, y_train, y_val = train_test_split(exog_train, endog_train, test_size=0.1, stratify=endog_train, random_state=42)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Model Build","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Random Forest","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Random Forest Model\n\nrf = RandomForestClassifier(random_state = 42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# Parameters used by the current forest\n\nprint('Parameters currently in use:\\n')\npprint(rf.get_params())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rf_base_model = RandomForestClassifier(n_estimators = 100, max_depth=5, random_state = 42)\nrf_base_model.fit(x_train, y_train)\nbase_accuracy = get_rocauc(rf_base_model, x_val, y_val)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Cross-Validation****","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# # Compute cross-validated AUC scores: cv_auc\n\n# cv_auc = cross_val_score(rf_base_model, x_val, y_val, cv=5, scoring = 'roc_auc')\n\n# print(\"AUC scores computed using 5-fold cross-validation: {}\".format(cv_auc))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### XGBoost Model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# fit model no training data\nxgb_model = XGBClassifier()\nxgb_model.fit(x_train, y_train)\n\n# # make predictions for test data\n# y_pred = xgb_model.predict(X_test)\nbase_accuracy_xgb = get_rocauc(xgb_model, x_val, y_val)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Cross-validation score","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"# Compute cross-validated AUC scores: cv_auc\n\nxgb_cv_auc = cross_val_score(xgb_model, x_val, y_val, cv=5, scoring = 'roc_auc')\n\nprint(\"AUC scores computed using 5-fold cross-validation: {}\".format(xgb_cv_auc))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Predictions","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"rf_pred_test= rf_base_model.predict_proba(exog_test)[:, 1]\n\nprint(rf_pred_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xgb_pred_test= xgb_model.predict_proba(exog_test)[:, 1]\n\nprint(xgb_pred_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# df_sub['target']=list(rf_pred_test)\ndf_sub['target']=list(xgb_pred_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_sub.to_csv( 'submission.csv', index=False )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Work in Progress !","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"#### Future Work:\n\n1. Include image data\n2. Tune model","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Upvote if you find this useful !","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}