{"cells":[{"metadata":{},"cell_type":"markdown","source":"# <font color = 'tomato'> Exploring and Sample Modelling on SIIM-ISIC Melanoma Data</font> ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Import Librires","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport plotly as py\nimport plotly.express as px\nimport plotly.graph_objs as go\nimport plotly.figure_factory as ff\nfrom plotly.offline import iplot, init_notebook_mode\n# Using plotly + cufflinks in offline mode\nimport cufflinks\ncufflinks.go_offline(connected=True)\ninit_notebook_mode(connected=True)\n\nfrom sklearn import preprocessing\n\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Read Data","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"path =  '/kaggle/input/siim-isic-melanoma-classification/'\n\ntrain = pd.read_csv(path+'train.csv')\n\ntest = pd.read_csv(path+'test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Filling NaN values in the data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train.sex.fillna('Not Provoded', inplace = True)\n\ntrain.age_approx.fillna(train.age_approx.mean(), inplace = True)\n\ntrain.anatom_site_general_challenge.fillna('UnKnown' , inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualise Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"b = train[train['target']==0]\nn = 16\nfig = plt.figure(figsize = (15,15))\nfor i, ind in zip(range(1, 1+n), [b.index[np.random.randint(b.shape[0])] for _ in range(n)]):\n    fig.add_subplot(4,4,i)   \n    plt.imshow(plt.imread(path+'jpeg/train/'+train.image_name[ind]+'.jpg'))\n    plt.axis('off')\n    plt.title('Benign'if train.target[ind] == 0 else 'Malignant')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"b = train[train['target']==1]\nn = 16\nfig = plt.figure(figsize = (15,15))\nfor i, ind in zip(range(1, 1+n), [b.index[np.random.randint(b.shape[0])] for _ in range(n)]):\n    fig.add_subplot(4,4,i)   \n    plt.imshow(plt.imread(path+'jpeg/train/'+train.image_name[ind]+'.jpg'))\n    plt.axis('off')\n    plt.title('Benign'if train.target[ind] == 0 else 'Malignant')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Few Insights\n## Which gender affected most?","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train.sex.value_counts()\nx = pd.DataFrame(data={'sex': x.index.tolist(), 'Count': x.values.tolist()})\nfig = px.pie(x, values='Count', names='sex', title='Gender Affected Most')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Age","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train.age_approx.value_counts()\n\ndf = pd.DataFrame({'Age':x.index, \n                  'Count':x.values})\npx.bar(df, x = 'Age', y = 'Count', color='Age', barmode='group')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Diagnosis","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train.diagnosis.value_counts()\nx = pd.DataFrame(data={'sex': x.index.tolist(), 'Count': x.values.tolist()})\nfig = px.pie(x, values='Count', names='sex', title='Gender Affected Most')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Modelling - XGBoost","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"tr = train[['sex', 'age_approx', 'anatom_site_general_challenge', 'target']]\ntr.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr.dtypes","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Encoding Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"label_encoder = preprocessing.LabelEncoder()\n\ntr['sex']= label_encoder.fit_transform(tr['sex']) \n\ntr['anatom_site_general_challenge']= label_encoder.fit_transform(tr['anatom_site_general_challenge']) \n\ntr.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.anatom_site_general_challenge.fillna('UnKnown' , inplace=True)\ntest.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ts = test[['sex', 'age_approx', 'anatom_site_general_challenge']]\n\nlabel_encoder = preprocessing.LabelEncoder()\n\nts['sex']= label_encoder.fit_transform(ts['sex']) \n\nts['anatom_site_general_challenge']= label_encoder.fit_transform(ts['anatom_site_general_challenge']) \n\nts.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(tr.iloc[:, :-1], tr.iloc[:, -1], test_size=0.3, random_state=11)\n\nmodel = XGBClassifier()\nmodel.fit(X_train, y_train)\n# make predictions for test data\ny_pred = model.predict(X_test)\npredictions = [round(value) for value in y_pred]\n# evaluate predictions\naccuracy = accuracy_score(y_test, predictions)\nprint(\"Accuracy: %.2f%%\" % (accuracy * 100.0))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(classification_report(y_test, predictions))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = confusion_matrix(y_test, predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ff.create_annotated_heatmap(\n    z=x,\n    x=[0,1],\n    y=[0,1],\n    annotation_text=x,\n    showscale=False, colorscale='Peach')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Predictions on unseen data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = model.predict(ts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.read_csv(path+'sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.target = pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.to_csv('submission_XGBoost.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Give an Upvote, If you like the work.","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}