{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt #for visualizations\nimport seaborn as sns #for interactive visualizations\n%matplotlib inline\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"785e39af71ea4ded7488a1cd93639a27d3e4c920"},"cell_type":"code","source":"#importing datasets\ntrain = pd.read_csv(\"../input/train.csv\")\ntest  = pd.read_csv(\"../input/test.csv\")\ntotaldata = [train, test]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#check both train and test datasets\ntrain.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e6ea9a33e1286618c4f9808c4db86b6424f49d7"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ace7996edbb3d7c3e1d8da089019c07a551c494e"},"cell_type":"code","source":"train.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"095d028465ec63ba4e06d4bdb1fd693601b9d104"},"cell_type":"code","source":"test.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed6901adcca4a11c61e6d7bd69b44ce5826c85e5"},"cell_type":"code","source":"#finding columns in train data set\ntrain.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"017f5fd814db84abdd4ec05d0463c914a67f1ac4"},"cell_type":"code","source":"#the above code gives you in Index form\n#this gives you in an array form\ntrain.columns.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"beff45363bbe9df845f5a3a1a545b1a5fbed994f"},"cell_type":"code","source":"train.columns[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8868aff2b0f37630b3df5c30c3722b6d78fdb66"},"cell_type":"code","source":"train.columns.values[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f505e00c70e80fd6b82746ccf31fb94d8012fda2"},"cell_type":"code","source":"#finding the shape(no.of.rows, no.of.columns) in train\ntrain.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7e2771143025785a65d722c956c4c0058223e29"},"cell_type":"code","source":"#finding the shape(no.of.rows, no.of. columns) in test\ntest.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c5a09191ed0bfe035c9b05acdf6770046be51a5"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0981ca55dfacff23ed9652cdb1eee8c1266b29c5"},"cell_type":"markdown","source":"Cabin and age has missing values, out of 12 coulmns 5 are object(text), 5 are int64(numbers) and 2 are float(in decimal points)\nName, Sex, Ticket, Cabin, Embarked are text data type.\nremaining are numericals.\nAge,Cabin  and Embarked has missing values."},{"metadata":{"trusted":true,"_uuid":"73d7278e8fbb72691714c979b8bd46bb97f646d2"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0181b491e8ca410ab705dd38245fd46af362c51c"},"cell_type":"markdown","source":"so we had missing values in Age,1 value in Fare and some values in Cabin"},{"metadata":{"trusted":true,"_uuid":"5e008678a1afb4b973244cf1418aea568e56d8b1"},"cell_type":"code","source":"train.describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fda80fda7074a90a33b8425cc3a5541732afebad"},"cell_type":"markdown","source":"**Data preprocessing **"},{"metadata":{"_uuid":"494f101c1221a94b9d6d648b16e38aa0c7ee5b4d"},"cell_type":"markdown","source":"handling missing values.\n\nfrom the train.info( ) we can say that we had missing values in Age, Cabin , Embarked columns as they have less rows out of 891.\n"},{"metadata":{"trusted":true,"_uuid":"58fe37af86a6e6942b25e4dd9372715cb3725c69"},"cell_type":"code","source":"#again checking missing values using isna() or isnull() method which most of the people do\ntrain.isna().sum\n#using this code snippet it only shows boolean values. if the column contains missing values then \n#it shows True, else it shows False.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26bafa3e3490d5691573abc1bb9b73ddb3ccc0dd"},"cell_type":"code","source":"train.isna().sum()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad73d703eb8f24864c8045f44422b48052e60dc0"},"cell_type":"code","source":"#same code as above but using isnull() method\ntrain.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bace63e8a274ba9aa5cfc1fce0d570089a07dc7d"},"cell_type":"code","source":"#from the above code snippet, its clear that we had missing values in Age, Cabin, Embarked columns.\n#Cabin has 687 missing values, Age has 177 missing values, Embarked has 2 missing values.\n#now try to fill those missing vlues.\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"95e4bc7c855be52c4aa62080f48b6f37b0121c4c"},"cell_type":"code","source":"test.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"061840708b0ff5a1683369d0716e2cb6b697051d"},"cell_type":"markdown","source":"We had 86 misssing values in Age, 327 missing values in Cabin, 1 in Fare"},{"metadata":{"trusted":true,"_uuid":"07282a37f3dc32fdd67c55b38d9f29f38edf08bf"},"cell_type":"code","source":"#lets replace missing values of Embarked column in trian data set\n#trying to find out what and how many are the values present in Embarked column\ntrain.Embarked.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4516e530b24c196dad23e51ba3f2a3e2f4d3faf7"},"cell_type":"code","source":"#mostly missing values are replaced with Either of Mean,Median,Mode\n#so totally we have 'S' repeated more times which is Mode case.\n#so fill the missing 2 values with most repeated value i.e \"S\"\ntrain[\"Embarked\"]  = train[\"Embarked\"].fillna(\"S\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48b751cceda7620b722575cbdef6e3082bbb09e2"},"cell_type":"code","source":"#after filling missing values in Embarked column, verifyinig still we have missing values in \n#Embarked column or not.\ntrain.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36af40263fc74df3fa411d187e7430b7e6152958"},"cell_type":"code","source":"#so now we do not have any missing values in Embarked columns. \n#fill missing value of Age with Median.in both test and train.\ntrain['Age'] = train['Age'].fillna(train['Age'].median())\ntest['Age'] = test['Age'].fillna(test['Age'].median())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"95f56157fc60be26dfdcec51d5a4a42b72855ac4"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3fb253c9d04683d1ead3685198a9044752a36b7f"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be5a58a24c9b07237dd8ec16c01b580fea5fe9b1"},"cell_type":"code","source":"#lets fill the 1 missing value in Fare column of test data set\ntest['Fare'] = test[\"Fare\"].fillna(\"0\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"600fd9a6401b94f73bebba3d551f49658b115409"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64868520a16b3b302cc2f506364d0c81d87c04d2"},"cell_type":"code","source":"#As Cabin Column has most data missing ,\n#there wont be any use filling missing values as it would be hard to fill those missing values using\n#using any of the techniques like Mean, Medain and  Mode.we have to delete it. \ntrain = train.drop(\"Cabin\",axis = 1)\ntest = test.drop(\"Cabin\", axis = 1)\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"591246eb337fb9198efe1e35d4747b881e274f16"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b7309dc804012344f7d46e55d8693dde85fee851"},"cell_type":"code","source":"#now we had another problem, i.e after filling all the missing values, now we have to convert text \n#data to numerical data, because that is what a machine understands at the end of the day.\n#for that we have to use label encoder\n#but before that lets find can we convert all the text values to numerical ones.\ntrain.Ticket.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fdf6dbdba7740ad95634fb8d49d7817524db1d24"},"cell_type":"code","source":"train['Ticket'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a312414fc80c44edf5f110ca61af49d7b4044baf"},"cell_type":"code","source":"#its clear that we had 681 unique values, so its impossible to encode it. so delete Ticket column\ntrain = train.drop(['Ticket'],axis = 1)\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"379bfd6a801af6b698539c7d78c16e6875e51c60"},"cell_type":"code","source":"test['Ticket'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cae59a8a64cc5f88eca655bbb0e15a7bfacb57e2"},"cell_type":"code","source":"test = test.drop(['Ticket'],axis = 1)\ntest.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d02d6c1a64ec062899a91cd012839c167cf27cc"},"cell_type":"code","source":"#also there is no use with name, so lets delete it\ntrain = train.drop([\"Name\"], axis = 1)\ntest = test.drop(['Name'], axis =1)\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b7df5957ae2d98cc7de252712bd436dab73f592"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1e3170a655030761889e18cbc5fec826efb96de"},"cell_type":"code","source":"#lets encode remaning 2 text columns to numerical ones\n#lets write function for label encoder\nfrom sklearn.preprocessing import LabelEncoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20a9e378822316308ce87283c5f656307a630ee1"},"cell_type":"code","source":"def encode_features(dataset,featurenames):\n    for featurename in featurenames:\n        LE = LabelEncoder()\n        LE.fit(dataset[featurename])\n        dataset[featurename] = LE.transform(dataset[featurename])\n    return dataset    \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d0b92ffa67a25b8fe7d206bf95102294f45ddf6"},"cell_type":"code","source":"\ntrain = encode_features(train, ['Sex','Embarked'])\ntest = encode_features(test, ['Sex','Embarked'])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d4a25f31a69e19eee591877e2549bd25674ba5b"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a8bd4b21f5308f646c8075834ce4758b008d03c"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b7e257724eabb8d7abd57599e37fab7af5785c5f"},"cell_type":"code","source":"#lets convert Fare from float to int\ntrain['Fare'] = train['Fare'].astype(int)\ntest[\"Fare\"] = test['Fare'].astype(int)\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ac7ac98b2dea2e5def37bf3fdee988a03c0610f"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ab61bca7d9ba07ac10c3fa9ef5a71ee2064a3416"},"cell_type":"markdown","source":"**Data visualization**"},{"metadata":{"_uuid":"690e8d7929988c4ee4187d67e24c3135fdb6cb65"},"cell_type":"markdown","source":"As our ML models fastens the processing if our data is in  0,1,2....., in simplie numeric form. so lets change Fare and Age to such format.\nnow we will find the correlation between Age and survival and between Fare and survival."},{"metadata":{"trusted":true,"_uuid":"223eec9b6173c00d1b1371de6f829c76518b6bf4"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c393601860f1e175f128033cad5a26a19cf4ce0"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0479143d7461187cd6a8d93e7c2822022b64143"},"cell_type":"code","source":"train = train.drop(['PassengerId'], axis=1)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0fe9fcfa52c9abc60adc02425b9564625eab23f0"},"cell_type":"markdown","source":"**ML models**"},{"metadata":{"trusted":true,"_uuid":"3cd1eb0b1ea652f3e0c5fb05c8872e8a6c096f7f"},"cell_type":"code","source":"X_train = train.drop(['Survived'], axis = 1)\nY_train = train[\"Survived\"]\nX_test  = test.drop(\"PassengerId\", axis=1).copy()\nX_train.shape, Y_train.shape, X_test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ad250688a223227d02e1bc0c9f122d64cbd0dec"},"cell_type":"code","source":"X_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77e629a545aae518c8ac289880bd5b50ccedf208"},"cell_type":"code","source":"Y_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"254bdd96c3c2e9a95f253673c14dece426647548"},"cell_type":"code","source":"X_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c89202674e90bda9bfb0a25aec52dbfbe612390"},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"958f9ae2690fae723af5c65f83cb71fb5fffb667"},"cell_type":"code","source":"#Logistic regression\nLR = LogisticRegression()\nLR.fit(X_train, Y_train)\ny_pred_LR = LR.predict(X_test)\nLR.score(X_train,Y_train)*100\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7403f0bdbc9f59d8d29d91a3747363fd1f9b699"},"cell_type":"code","source":"#Support vector machine \nsvm = SVC()\nsvm.fit(X_train, Y_train)\ny_pred_svm = svm.predict(X_test)\nsvm.score(X_train,Y_train)*100","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44465a31f0f0fa6b784f2bdec4b7b8ff211f875e"},"cell_type":"code","source":"#RandomForestClassirier\nrfc = RandomForestClassifier()\nrfc.fit(X_train, Y_train)\ny_pred_rfc = rfc.predict(X_test)\nrfc.score(X_train,Y_train)*100","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8601018c55d3713b4f284fa7e4cc9aa7baf8b7dd"},"cell_type":"code","source":"#KNeighborsClassifier\nknc = KNeighborsClassifier()\nknc.fit(X_train, Y_train)\ny_pred_knc = knc.predict(X_test)\nknc.score(X_train,Y_train)*100","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b899a1b1ad9d764370391cf99f0e9f072446a04e"},"cell_type":"code","source":"#KNeighborsClassifier\ndtc = DecisionTreeClassifier()\ndtc.fit(X_train, Y_train)\ny_pred_dtc = dtc.predict(X_test)\ndtc.score(X_train,Y_train)*100","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2a3cb632b010b56e2d405479752887c4d84c317c"},"cell_type":"markdown","source":"**If you find any mistakes or if you have any suggestions, please comment down. I would like to learn from my mistakes.**"},{"metadata":{"trusted":true,"_uuid":"071bce79025a0ae4986c2448ef60a2e34e781d66"},"cell_type":"code","source":"submission = pd.DataFrame({\n        \"PassengerId\": test[\"PassengerId\"],\n        \"Survived\": y_pred_dtc\n    })\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66d2f3a9cd984ab789f98244970d237f4cde415e"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf03d25ace77ea6574c585c250068c0db0f51129"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}