{"cells":[{"metadata":{},"cell_type":"markdown","source":"<h1>Objective</h1>\nThe target of this notebook is to aid with the start of the development, it contains:<br>\n* Data loading\n* Graphic visualization of the distribution of columns\n* Missing values removal\n* Visualization of sample random values\n<br>\n\nAll the typical initial preprocessing of data.<br>Hope it can help somebody, if you find it usefull or like the idea a newbie trying to aid anybody, please upvote me.<br>\nRegards.","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\n#for dirname, _, filenames in os.walk('../input/siim-isic-melanoma-classification/jpeg/'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nsns.set()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* <h1>Loading Datasets</h1>\nLoading of train and test files and basic display of column information and distribution.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_DIR = '../input/siim-isic-melanoma-classification/'\ntrain = pd.read_csv(BASE_DIR + 'train.csv')\ntest  = pd.read_csv(BASE_DIR + 'test.csv')\nim_train_path = BASE_DIR + 'jpeg/train/'\nim_test_path = BASE_DIR + 'jpeg/test/'\n\ndef dataPlot(train, train_malignant):\n    train_malignant = train[train['benign_malignant']=='malignant']\n    targets = ['sex', 'age_approx', 'benign_malignant','anatom_site_general_challenge','diagnosis']\n    fig, ax = plt.subplots(3, 2, figsize=(20, 10))\n    sns.countplot(x=targets[0], data=train, ax=ax[0, 0])\n    sns.countplot(x=targets[1], data=train, ax=ax[0, 1])\n    sns.countplot(x=targets[2], data=train, ax=ax[1, 0])\n    sns.countplot(x=targets[3], data=train, ax=ax[1, 1])\n    sns.countplot(x=targets[4], data=train_malignant, ax=ax[2, 0])\n    plt.show()\n    return\n\n#Data review\nprint('Train shape and sample')\ntrain.shape\nprint(train.sample(n=5, random_state=1))\nprint('Test shape and sample')\ntest.shape\nprint(train.sample(n=3, random_state=1))\nprint('Sex:\\n',train['sex'].value_counts())\nprint('Age:\\n',train['age_approx'].value_counts())\nprint('Localization:\\n',train['anatom_site_general_challenge'].value_counts())\nprint(train['benign_malignant'].value_counts())\ntrain_malignant = train[train['benign_malignant']=='malignant']\ndataPlot(train, train_malignant)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<h1>Missing values</h1>\nDisplay missing information (null or missing values)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def missing_values_table(df):\n        print('Size of table is ',df.size)\n        mis_val = df.isnull().sum()\n        mis_val_percent = 100 * df.isnull().sum() / len(df)\n        mis_val_table = pd.concat([mis_val, mis_val_percent], axis=1)\n        mis_val_table_ren_columns = mis_val_table.rename(\n        columns = {0 : 'Missing Values', 1 : '% of Total Values'})\n        mis_val_table_ren_columns = mis_val_table_ren_columns[\n            mis_val_table_ren_columns.iloc[:,1] != 0].sort_values(\n        '% of Total Values', ascending=False).round(1)\n        print (\"Your selected dataframe has \" + str(df.shape[1]) + \" columns.\\n\"      \n            \"There are \" + str(mis_val_table_ren_columns.shape[0]) +\n              \" columns that have missing values.\")\n        return mis_val_table_ren_columns\n    \nprint('Train data missing (null or NaN values)')\nprint(missing_values_table(train))\nprint('\\nTest data missing (null or NaN values)')\nprint(missing_values_table(test))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Missing values are only a very small percentage of samples, we'll remove them form the dataset.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fill = train.fillna('unknown')\ntest_fill = test.fillna('unknown')\n#Remove rows  with unknown values based in previous work\ntrain_fill = train_fill[train_fill.anatom_site_general_challenge != 'unknown']\ntrain_fill = train_fill[train_fill.age_approx != 'unknown']\ntrain_fill = train_fill[train_fill.sex != 'unknown']\ntest_fill = test_fill[test_fill.anatom_site_general_challenge != 'unknown']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Visualization of sample data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image\ndef drawImages(sample):\n    plt.figure(figsize = (18,18))\n    for iterator, filename in enumerate(sample):\n        image = Image.open(filename)\n        plt.subplot(4,2,iterator+1)\n        plt.imshow(image)   \n    plt.tight_layout()\n    return\n\nprint(\"Sample benign images\")\nsample_train = im_train_path + train_fill[train_fill['benign_malignant']=='benign'].sample(4).image_name + '.jpg'\ndrawImages(sample_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Sample malignant images\")\nsample_train = im_train_path + train_fill[train_fill['benign_malignant']=='malignant'].sample(4).image_name + '.jpg'\ndrawImages(sample_train)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}