{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# read csv files\ntrain = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntrain","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# understand the train dataset\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### **Total number of records: 33126**\n\n\n\n***Therefore columns with null entries are:***\n* sex\n* age_approx\n* anatom_site_general_challenge  ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## 1. Sex:\nlets fill in with either male or female according to which gender is more common","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#find the ratio of men to total ratio:\nno_of_male = (train['sex'] == 'male').sum()\nno_of_values = (train['sex'].notna()).count()\nprint(no_of_male/no_of_values)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"since the ratio of males > 0.5, let's fill na values with male","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train['sex'] = train['sex'].fillna('male')\nprint('no. of na values in sex = ', train['sex'].isna().sum() )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 2. age_approx:\n\nlet's follow the best practice and replace na values with median age","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train['age_approx'] = train['age_approx'].fillna( train['age_approx'].median() )\n\nprint('no. of na values in age_approx = ', train['age_approx'].isna().sum() )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 3. anatom_site_general_challenge:\nfill it with the most common site as it is the mst probable !","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"most_freq_site = train['anatom_site_general_challenge'].value_counts().idxmax()\n\ntrain['anatom_site_general_challenge'].fillna( most_freq_site, inplace = True)\n\nprint('no. of na values in anatom_site_general_challenge = ', train['anatom_site_general_challenge'].isna().sum() )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Diagnosis:\nThis column will not be available in the test dataset, so we cannot use it for prediction directly but only for EDA, so let's drop the Diagnosis column for now","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train.drop('diagnosis', axis = 1, inplace = True)\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets clean this table for data analysis\n\n# 1. Lets create a separate dataframe for this purpose\np_data = train.copy()  # p_data as in person data\n\n# 2. Lets drop the columns image_name and patient_id\n#         we'll access the image_name from the train dataset \n#         and since patient_id doens't represent any probability of being malignant or benign, \n#                     - we'll treat column id as new patient id\n\np_data.drop(['image_name', 'patient_id'],axis = 1, inplace = True)\n\n# if you want to preserve patient_id to be the id instead use:\n\n# p_data.drop('image_name',axis = 1, inplace = True)\n# p_data.set_index('patient_id')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## lets convert all categorical data into bool type!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# convert sex into a bool column\np_data['is_male'] = (p_data['sex'] == 'male')\np_data.drop('sex', axis = 1, inplace = True)\n\n#convert benign_malignant into a bool column'\np_data['is_malignant'] = p_data['target']\np_data.drop(['benign_malignant', 'target'], axis = 1, inplace = True)\n\n\np_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#convert anatom_site_general_challenge into categories\n\n#note: we inclued all the sites except for the final site cause in the final dataframe if we have false for\n#all the included sites it means true automatically for the last site so adding it will only ceate redundancy\n\nregions = p_data['anatom_site_general_challenge'].unique()[:-1]\n\nfor region in regions:\n    p_data['is_' + region] = (p_data['anatom_site_general_challenge'] == region)\n    print(region)\n\np_data.drop('anatom_site_general_challenge', axis = 1,  inplace = True)\np_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# normalize age\np_data['normalized_age'] = p_data['age_approx']/p_data['age_approx'].max()\np_data.drop('age_approx', axis = 1, inplace = True)\n\np_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}