{"cells":[{"metadata":{},"cell_type":"markdown","source":"## My first Kaggle kernel. Your suggestions and comments will help me improve. ##","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\nimport cv2\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\")\ndf_test  = pd.read_csv(\"../input/siim-isic-melanoma-classification/test.csv\")\n\nprint(\"Train shape:\", df_train.shape)\nprint(\"Test shape:\", df_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.sample(5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Test set doesn't have diagnosis and benign_malignant columns.**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_patient_train = df_train.patient_id.unique()\nunique_patient_test = df_test.patient_id.unique()\n\nprint(\"Unique patients in training set:\", unique_patient_train.shape[0])\nprint(\"Unique patients in test set:\", unique_patient_test.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"No of patients common in train and test sets:\", np.intersect1d(unique_patient_train, unique_patient_test).shape[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Since the test set doesn't have diagnosis and benign_malignant, let's see what's there on those two columns.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.diagnosis.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Set the value of \"unknown\" as nan\ndf_train.diagnosis = df_train.diagnosis.apply(lambda x: np.nan if x == \"unknown\" else x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.diagnosis.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.benign_malignant.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.benign_malignant.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.target.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = sns.countplot(x = \"benign_malignant\", hue = \"target\", data = df_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"So the benign has been encoded as target = 0 and malignant has been encoded as target = 1. So we can safely remove the column benign_malignant.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.drop(\"benign_malignant\", axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's drop the diagnosis as well as it has many null entry and is not present in the test set\ndf_train.drop(\"diagnosis\", axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Train shape:\", df_train.shape)\nprint(\"Test shape:\", df_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.isnull().sum().to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.isnull().sum().to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(df_train.sex.value_counts() * 100 / df_train.shape[0]).to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(df_test.sex.value_counts() * 100 / df_test.shape[0]).to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Since male and female ratio is almost same, with mens are more in number, let's fill those 65 missing items as male\ndf_train.sex = df_train.sex.fillna(\"male\")\n(df_train.sex.value_counts() * 100 / df_train.shape[0]).to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = sns.countplot(x = \"sex\", hue = \"target\", data = df_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# For age_approx, fill the na values with 0.0\ndf_train.age_approx = df_train.age_approx.fillna(0.0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize = (10, 5))\nax = sns.countplot(x = \"age_approx\", hue = \"target\", data = df_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(df_train.anatom_site_general_challenge.value_counts() * 100 / df_train.shape[0]).to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(df_test.anatom_site_general_challenge.value_counts() * 100 / df_test.shape[0]).to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# anatom_site_general_challenge needs to be imputed for both train and test. Let's use something called unknown for now\ndf_train.anatom_site_general_challenge = df_train.anatom_site_general_challenge.fillna(\"unknown\")\ndf_test.anatom_site_general_challenge = df_test.anatom_site_general_challenge.fillna(\"unknown\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's do one hot encoding on sex column\ndf_train[\"sex_cat_m\"] = 0\ndf_train.loc[df_train.sex == \"male\", \"sex_cat_m\"] = 1\ndf_train.drop(\"sex\", axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's do one hot encoding on anatom_site_general_challenge column\n# torso, lower extremity, upper extremity, head/neck, unknown, palms/soles, oral/genital\n\ntemp = pd.get_dummies(df_train.anatom_site_general_challenge, prefix = \"location\")\ntemp.drop(\"location_unknown\", axis = 1, inplace = True)\ntemp.columns = [(\"location_cat_\" + str(i)) for i in range(1, 7)]\n\ndf_train = pd.concat([df_train, temp], axis = 1)\ndf_train.drop(\"anatom_site_general_challenge\", axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp = pd.get_dummies(df_test.anatom_site_general_challenge, prefix = \"location\")\ntemp.drop(\"location_unknown\", axis = 1, inplace = True)\ntemp.columns = [(\"location_cat_\" + str(i)) for i in range(1, 7)]\n\ndf_test = pd.concat([df_test, temp], axis = 1)\ndf_test.drop(\"anatom_site_general_challenge\", axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"values = df_train.target.values\ndf_train.drop(\"target\", axis = 1, inplace = True)\ndf_train[\"target\"] = values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Train shape:\", df_train.shape)\nprint(\"Test shape:\", df_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.corr().style.background_gradient(cmap = \"RdBu\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Somehow can we impute the missing entries?**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.target.value_counts().to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(df_train.target.value_counts() * 100 / df_train.shape[0]).to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"The dataset is very much imbalanced.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"**How does the images looks like?**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_images(df, n_rows = 5, n_cols = 5, figsize = (20, 20), resize = (1024, 1024), preprocessing = None, label = 0):\n    query_string = \"target == {}\".format(label)\n    df = df.query(query_string).reset_index(drop = True)\n    fig = plt.figure(figsize = figsize)\n    ax  = []\n    base_path = \"../input/siim-isic-melanoma-classification/jpeg/train/\"\n\n    for i in range(n_rows * n_cols):\n        img = plt.imread(base_path + df.loc[i, \"image_name\"] + \".jpg\")\n        img = cv2.resize(img, resize)\n\n        if preprocessing:\n            img = preprocessing(img)\n\n        ax.append(fig.add_subplot(n_rows, n_cols, i + 1) )\n        plot_title = \"Image {}: {}\".format(str(i + 1), \"Benign\" if label == 0 else \"Malignant\") \n        ax[-1].set_title(plot_title)\n        plt.imshow(img, alpha = 1, cmap = \"gray\")\n\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Training images - Benign\nplot_images(df_train, label = 0, resize = (224, 224))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Training images - Malignant\nplot_images(df_train, label = 1, resize = (224, 224))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"#### Observations: ####\n - Dataset is very much imbalanced. Need external data for malignant cases.\n - anatom_site_general_challenge, age and sex need to be imputed and make those part of split/model.\n - In training set, there are 33126 images where there are 2056 unique patients. So on an average, there are around 15 images per patient. So while splitting the dataset for CV, need to take care of this fact.\n - Scaling the images as 224 x 224, instead of 1024 x 1024 can be done.","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}