{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T15:25:30.921021Z","iopub.execute_input":"2022-07-09T15:25:30.921724Z","iopub.status.idle":"2022-07-09T15:25:30.932240Z","shell.execute_reply.started":"2022-07-09T15:25:30.921677Z","shell.execute_reply":"2022-07-09T15:25:30.931338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Import Dependencies","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:30.936464Z","iopub.execute_input":"2022-07-09T15:25:30.937372Z","iopub.status.idle":"2022-07-09T15:25:30.950686Z","shell.execute_reply.started":"2022-07-09T15:25:30.937308Z","shell.execute_reply":"2022-07-09T15:25:30.949833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Import Data","metadata":{}},{"cell_type":"code","source":"dataset = pd.read_csv(\"../input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:30.952154Z","iopub.execute_input":"2022-07-09T15:25:30.952527Z","iopub.status.idle":"2022-07-09T15:25:30.969420Z","shell.execute_reply.started":"2022-07-09T15:25:30.952495Z","shell.execute_reply":"2022-07-09T15:25:30.968271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Analyze Data","metadata":{}},{"cell_type":"code","source":"dataset.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:30.971062Z","iopub.execute_input":"2022-07-09T15:25:30.973251Z","iopub.status.idle":"2022-07-09T15:25:31.006546Z","shell.execute_reply.started":"2022-07-09T15:25:30.973200Z","shell.execute_reply":"2022-07-09T15:25:31.005193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.009141Z","iopub.execute_input":"2022-07-09T15:25:31.009744Z","iopub.status.idle":"2022-07-09T15:25:31.023692Z","shell.execute_reply.started":"2022-07-09T15:25:31.009707Z","shell.execute_reply":"2022-07-09T15:25:31.022644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Divide dataset into Dependent and Independent Variable","metadata":{}},{"cell_type":"code","source":"X_train = dataset.iloc[: , 2 : ]\ny_train = dataset.iloc[: , 1]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.025596Z","iopub.execute_input":"2022-07-09T15:25:31.026310Z","iopub.status.idle":"2022-07-09T15:25:31.032663Z","shell.execute_reply.started":"2022-07-09T15:25:31.026261Z","shell.execute_reply":"2022-07-09T15:25:31.031776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here , PassengerId will not affect the dependent variable . So We do not consider it to be part of training data","metadata":{}},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.034377Z","iopub.execute_input":"2022-07-09T15:25:31.034720Z","iopub.status.idle":"2022-07-09T15:25:31.068912Z","shell.execute_reply.started":"2022-07-09T15:25:31.034689Z","shell.execute_reply":"2022-07-09T15:25:31.068132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.070139Z","iopub.execute_input":"2022-07-09T15:25:31.070593Z","iopub.status.idle":"2022-07-09T15:25:31.077765Z","shell.execute_reply.started":"2022-07-09T15:25:31.070564Z","shell.execute_reply":"2022-07-09T15:25:31.076743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Columns","metadata":{}},{"cell_type":"code","source":"dataset.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.079849Z","iopub.execute_input":"2022-07-09T15:25:31.080379Z","iopub.status.idle":"2022-07-09T15:25:31.091751Z","shell.execute_reply.started":"2022-07-09T15:25:31.080351Z","shell.execute_reply":"2022-07-09T15:25:31.090989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PClass","metadata":{}},{"cell_type":"code","source":"Pclass = dataset['Pclass']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.092745Z","iopub.execute_input":"2022-07-09T15:25:31.093369Z","iopub.status.idle":"2022-07-09T15:25:31.103442Z","shell.execute_reply.started":"2022-07-09T15:25:31.093322Z","shell.execute_reply":"2022-07-09T15:25:31.101951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Pclass","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.105152Z","iopub.execute_input":"2022-07-09T15:25:31.106231Z","iopub.status.idle":"2022-07-09T15:25:31.118340Z","shell.execute_reply.started":"2022-07-09T15:25:31.106186Z","shell.execute_reply":"2022-07-09T15:25:31.117481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Analyze PClass","metadata":{}},{"cell_type":"code","source":"Pclass.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.120441Z","iopub.execute_input":"2022-07-09T15:25:31.120966Z","iopub.status.idle":"2022-07-09T15:25:31.132901Z","shell.execute_reply.started":"2022-07-09T15:25:31.120915Z","shell.execute_reply":"2022-07-09T15:25:31.132101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Pclass is more skewed to 3","metadata":{}},{"cell_type":"code","source":"plt.hist(Pclass)\nplt.title('PClass')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.134580Z","iopub.execute_input":"2022-07-09T15:25:31.135052Z","iopub.status.idle":"2022-07-09T15:25:31.341066Z","shell.execute_reply.started":"2022-07-09T15:25:31.135017Z","shell.execute_reply":"2022-07-09T15:25:31.339968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.boxplot(Pclass)\nplt.title('PClass')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.342298Z","iopub.execute_input":"2022-07-09T15:25:31.342584Z","iopub.status.idle":"2022-07-09T15:25:31.510657Z","shell.execute_reply.started":"2022-07-09T15:25:31.342557Z","shell.execute_reply":"2022-07-09T15:25:31.509947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above Boxplot , we observe that majority of passengers belong to third class","metadata":{}},{"cell_type":"markdown","source":"Observing Dependency","metadata":{}},{"cell_type":"code","source":"plt.scatter(Pclass.values , y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.512903Z","iopub.execute_input":"2022-07-09T15:25:31.513479Z","iopub.status.idle":"2022-07-09T15:25:31.707803Z","shell.execute_reply.started":"2022-07-09T15:25:31.513443Z","shell.execute_reply":"2022-07-09T15:25:31.706740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since , PClass has all values as non null , and does not have much of outliers , there is no need to do data preprocessing for PClass","metadata":{}},{"cell_type":"markdown","source":"Name\nName of a passenger does not affect if they survived or not . \nHence , the Class Name is dropped from the Independent Variable","metadata":{}},{"cell_type":"code","source":"dataset.drop(['Name'] , inplace = True , axis =1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.708992Z","iopub.execute_input":"2022-07-09T15:25:31.709303Z","iopub.status.idle":"2022-07-09T15:25:31.715611Z","shell.execute_reply.started":"2022-07-09T15:25:31.709274Z","shell.execute_reply":"2022-07-09T15:25:31.714471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sex","metadata":{}},{"cell_type":"markdown","source":"Since , female passengers where given first preference  , a passenger's gender will make a big difference","metadata":{}},{"cell_type":"code","source":"sex = dataset['Sex']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.716911Z","iopub.execute_input":"2022-07-09T15:25:31.717238Z","iopub.status.idle":"2022-07-09T15:25:31.727032Z","shell.execute_reply.started":"2022-07-09T15:25:31.717208Z","shell.execute_reply":"2022-07-09T15:25:31.726021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sex","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.727965Z","iopub.execute_input":"2022-07-09T15:25:31.729203Z","iopub.status.idle":"2022-07-09T15:25:31.741231Z","shell.execute_reply.started":"2022-07-09T15:25:31.729167Z","shell.execute_reply":"2022-07-09T15:25:31.740086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For our convenience , let us set 'male' as 1 , and 'female' as 0","metadata":{}},{"cell_type":"code","source":"dataset['Sex'] = dataset['Sex'].replace('male' , 1)\ndataset['Sex'] = dataset['Sex'].replace('female' , 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.743060Z","iopub.execute_input":"2022-07-09T15:25:31.743767Z","iopub.status.idle":"2022-07-09T15:25:31.751905Z","shell.execute_reply.started":"2022-07-09T15:25:31.743709Z","shell.execute_reply":"2022-07-09T15:25:31.751131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Sex'].describe()\nsex = dataset['Sex']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.753305Z","iopub.execute_input":"2022-07-09T15:25:31.754115Z","iopub.status.idle":"2022-07-09T15:25:31.769086Z","shell.execute_reply.started":"2022-07-09T15:25:31.754067Z","shell.execute_reply":"2022-07-09T15:25:31.767897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualise Data","metadata":{}},{"cell_type":"code","source":"plt.hist(sex )\nplt.title('Sex')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.770473Z","iopub.execute_input":"2022-07-09T15:25:31.770917Z","iopub.status.idle":"2022-07-09T15:25:31.969088Z","shell.execute_reply.started":"2022-07-09T15:25:31.770881Z","shell.execute_reply":"2022-07-09T15:25:31.967889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age","metadata":{}},{"cell_type":"markdown","source":"A person's age mattered whether he or she  could have survived or not . \nAs children are given greater preference than adults , lower ages must be given more prefrence","metadata":{}},{"cell_type":"code","source":"age = dataset['Age']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.970624Z","iopub.execute_input":"2022-07-09T15:25:31.971067Z","iopub.status.idle":"2022-07-09T15:25:31.978076Z","shell.execute_reply.started":"2022-07-09T15:25:31.971020Z","shell.execute_reply":"2022-07-09T15:25:31.976993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.982761Z","iopub.execute_input":"2022-07-09T15:25:31.983515Z","iopub.status.idle":"2022-07-09T15:25:31.993583Z","shell.execute_reply.started":"2022-07-09T15:25:31.983445Z","shell.execute_reply":"2022-07-09T15:25:31.992601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Age collumn has many nan values . So we should try to replace the nan values","metadata":{}},{"cell_type":"code","source":"pd.DataFrame(dataset['Age']).info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:31.994809Z","iopub.execute_input":"2022-07-09T15:25:31.995530Z","iopub.status.idle":"2022-07-09T15:25:32.010582Z","shell.execute_reply.started":"2022-07-09T15:25:31.995493Z","shell.execute_reply":"2022-07-09T15:25:32.009731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(dataset['Age']).describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:32.012248Z","iopub.execute_input":"2022-07-09T15:25:32.012720Z","iopub.status.idle":"2022-07-09T15:25:32.034464Z","shell.execute_reply.started":"2022-07-09T15:25:32.012674Z","shell.execute_reply":"2022-07-09T15:25:32.033635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agedf = pd.DataFrame(age)\nnon_age_df = agedf.dropna()\nplt.hist(non_age_df.values , bins = 80)\nplt.title('Age')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:32.035615Z","iopub.execute_input":"2022-07-09T15:25:32.036168Z","iopub.status.idle":"2022-07-09T15:25:32.366387Z","shell.execute_reply.started":"2022-07-09T15:25:32.036134Z","shell.execute_reply":"2022-07-09T15:25:32.364980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.boxplot(non_age_df.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:32.367990Z","iopub.execute_input":"2022-07-09T15:25:32.368342Z","iopub.status.idle":"2022-07-09T15:25:32.534638Z","shell.execute_reply.started":"2022-07-09T15:25:32.368310Z","shell.execute_reply":"2022-07-09T15:25:32.533506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Outlier Treatment ","metadata":{}},{"cell_type":"code","source":"Q1 = non_age_df.quantile(0.25)\nQ2 = non_age_df.quantile(0.50)\nQ3 = non_age_df.quantile(0.75)\nIQR = Q3 - Q1\nmin = Q1 - (1.5*IQR)\nmax = Q3 + (1.5*IQR)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:32.535881Z","iopub.execute_input":"2022-07-09T15:25:32.536870Z","iopub.status.idle":"2022-07-09T15:25:32.547208Z","shell.execute_reply.started":"2022-07-09T15:25:32.536830Z","shell.execute_reply":"2022-07-09T15:25:32.546184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"First Quartile : \" , Q1)\nprint(\"Second Quartile : \" , Q2)\nprint(\"Third Quartile : \" , Q3)\nprint(\"Inter Quartile Range : \" , IQR)\nprint(\"Maximum Value preferred to not be an outlier  : \" , min)\nprint(\"Minimum Value preferred to not be an outlier  : \" , max)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:32.548506Z","iopub.execute_input":"2022-07-09T15:25:32.548854Z","iopub.status.idle":"2022-07-09T15:25:32.567819Z","shell.execute_reply.started":"2022-07-09T15:25:32.548815Z","shell.execute_reply":"2022-07-09T15:25:32.566769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Determine number of outliers","metadata":{}},{"cell_type":"code","source":"outliers = dataset[(agedf>max) |(agedf<min) ].index\nprint(\"Number of Outliers : \" , len(outliers))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:32.569246Z","iopub.execute_input":"2022-07-09T15:25:32.569562Z","iopub.status.idle":"2022-07-09T15:25:32.589473Z","shell.execute_reply.started":"2022-07-09T15:25:32.569533Z","shell.execute_reply":"2022-07-09T15:25:32.588373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is better if we replace the outliers with a suitable value , along with np.nan values","metadata":{}},{"cell_type":"code","source":"plt.hist(non_age_df )","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:32.591224Z","iopub.execute_input":"2022-07-09T15:25:32.592007Z","iopub.status.idle":"2022-07-09T15:25:32.804694Z","shell.execute_reply.started":"2022-07-09T15:25:32.591958Z","shell.execute_reply":"2022-07-09T15:25:32.803641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Replace outliers with median of non_age_values","metadata":{}},{"cell_type":"code","source":"median = non_age_df.quantile(0.5)\nfor i in list(outliers):\n    dataset.iloc[i , 4] = median","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:32.806314Z","iopub.execute_input":"2022-07-09T15:25:32.806626Z","iopub.status.idle":"2022-07-09T15:25:33.014333Z","shell.execute_reply.started":"2022-07-09T15:25:32.806598Z","shell.execute_reply":"2022-07-09T15:25:33.013233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Replace nan values by median","metadata":{}},{"cell_type":"code","source":"age_data = pd.DataFrame(dataset['Age'])\ndataset['Age'] = age_data.replace(np.nan , median)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.016183Z","iopub.execute_input":"2022-07-09T15:25:33.016509Z","iopub.status.idle":"2022-07-09T15:25:33.024893Z","shell.execute_reply.started":"2022-07-09T15:25:33.016477Z","shell.execute_reply":"2022-07-09T15:25:33.023602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualise new age data","metadata":{}},{"cell_type":"code","source":"age = pd.DataFrame(dataset['Age'])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.026298Z","iopub.execute_input":"2022-07-09T15:25:33.026597Z","iopub.status.idle":"2022-07-09T15:25:33.036268Z","shell.execute_reply.started":"2022-07-09T15:25:33.026570Z","shell.execute_reply":"2022-07-09T15:25:33.035413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.037427Z","iopub.execute_input":"2022-07-09T15:25:33.037736Z","iopub.status.idle":"2022-07-09T15:25:33.052518Z","shell.execute_reply.started":"2022-07-09T15:25:33.037707Z","shell.execute_reply":"2022-07-09T15:25:33.051361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age has now been preprocessed . Let us confirm this once ","metadata":{}},{"cell_type":"code","source":"plt.hist(age)\nplt.title('New age')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.053761Z","iopub.execute_input":"2022-07-09T15:25:33.054218Z","iopub.status.idle":"2022-07-09T15:25:33.240235Z","shell.execute_reply.started":"2022-07-09T15:25:33.054184Z","shell.execute_reply":"2022-07-09T15:25:33.238984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age.skew()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.241825Z","iopub.execute_input":"2022-07-09T15:25:33.242828Z","iopub.status.idle":"2022-07-09T15:25:33.253550Z","shell.execute_reply.started":"2022-07-09T15:25:33.242777Z","shell.execute_reply":"2022-07-09T15:25:33.252213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now age skewness is near to 0 . This is a good Sign","metadata":{}},{"cell_type":"markdown","source":"SibSp","metadata":{}},{"cell_type":"code","source":"Sibsp = dataset['SibSp']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.254854Z","iopub.execute_input":"2022-07-09T15:25:33.255831Z","iopub.status.idle":"2022-07-09T15:25:33.261008Z","shell.execute_reply.started":"2022-07-09T15:25:33.255792Z","shell.execute_reply":"2022-07-09T15:25:33.259718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sibsp.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.263118Z","iopub.execute_input":"2022-07-09T15:25:33.263897Z","iopub.status.idle":"2022-07-09T15:25:33.276347Z","shell.execute_reply.started":"2022-07-09T15:25:33.263849Z","shell.execute_reply":"2022-07-09T15:25:33.275264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visual Data","metadata":{}},{"cell_type":"code","source":"plt.hist(Sibsp)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.277494Z","iopub.execute_input":"2022-07-09T15:25:33.278259Z","iopub.status.idle":"2022-07-09T15:25:33.480620Z","shell.execute_reply.started":"2022-07-09T15:25:33.278226Z","shell.execute_reply":"2022-07-09T15:25:33.479560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.boxplot(Sibsp)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.482247Z","iopub.execute_input":"2022-07-09T15:25:33.482828Z","iopub.status.idle":"2022-07-09T15:25:33.647172Z","shell.execute_reply.started":"2022-07-09T15:25:33.482791Z","shell.execute_reply":"2022-07-09T15:25:33.646273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that # of siblings or spouses , has less number of outliers . So , we can replace the outliers by the 0 , as 0 occurs most . Note that this must be an integer only","metadata":{}},{"cell_type":"code","source":"Q1 = Sibsp.quantile(0.25)\nQ2 = Sibsp.quantile(0.50)\nQ3 = Sibsp.quantile(0.75)\nIQR = Q3 - Q1\nmin = Q1 - (1.5*IQR)\nmax = Q3 + (1.5*IQR)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.649089Z","iopub.execute_input":"2022-07-09T15:25:33.650070Z","iopub.status.idle":"2022-07-09T15:25:33.661079Z","shell.execute_reply.started":"2022-07-09T15:25:33.650020Z","shell.execute_reply":"2022-07-09T15:25:33.660001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"First Quartile is : \" , Q1)\nprint(\"Second Quartile is : \" , Q2)\nprint(\"Third Quartile is : \" , Q3)\nprint(\"IQR is : \" , IQR)\nprint(\"Minimum Value to not be an outlier is  : \" , min)\nprint(\"Maximum Value to not be an outlier is  :\" , max)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.662385Z","iopub.execute_input":"2022-07-09T15:25:33.663297Z","iopub.status.idle":"2022-07-09T15:25:33.671573Z","shell.execute_reply.started":"2022-07-09T15:25:33.663258Z","shell.execute_reply":"2022-07-09T15:25:33.670466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Determine the outliers ","metadata":{}},{"cell_type":"code","source":"outliers = dataset[(dataset['SibSp']>max) |(dataset['SibSp']<min) ].index","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.673080Z","iopub.execute_input":"2022-07-09T15:25:33.675442Z","iopub.status.idle":"2022-07-09T15:25:33.684033Z","shell.execute_reply.started":"2022-07-09T15:25:33.675393Z","shell.execute_reply":"2022-07-09T15:25:33.682993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in outliers:\n    dataset.loc[i , 'SibSp'] = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.685444Z","iopub.execute_input":"2022-07-09T15:25:33.686107Z","iopub.status.idle":"2022-07-09T15:25:33.705893Z","shell.execute_reply.started":"2022-07-09T15:25:33.686063Z","shell.execute_reply":"2022-07-09T15:25:33.704993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.707298Z","iopub.execute_input":"2022-07-09T15:25:33.708347Z","iopub.status.idle":"2022-07-09T15:25:33.722239Z","shell.execute_reply.started":"2022-07-09T15:25:33.708289Z","shell.execute_reply":"2022-07-09T15:25:33.721340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Parch . Number of parents / children aboard the Titanic","metadata":{}},{"cell_type":"code","source":"Parch = dataset['Parch']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.733731Z","iopub.execute_input":"2022-07-09T15:25:33.734445Z","iopub.status.idle":"2022-07-09T15:25:33.739154Z","shell.execute_reply.started":"2022-07-09T15:25:33.734406Z","shell.execute_reply":"2022-07-09T15:25:33.737797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Parch.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.740368Z","iopub.execute_input":"2022-07-09T15:25:33.740682Z","iopub.status.idle":"2022-07-09T15:25:33.757991Z","shell.execute_reply.started":"2022-07-09T15:25:33.740654Z","shell.execute_reply":"2022-07-09T15:25:33.756535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This has many outliers","metadata":{}},{"cell_type":"code","source":"plt.hist(Parch)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.759701Z","iopub.execute_input":"2022-07-09T15:25:33.760164Z","iopub.status.idle":"2022-07-09T15:25:33.966084Z","shell.execute_reply.started":"2022-07-09T15:25:33.760120Z","shell.execute_reply":"2022-07-09T15:25:33.965032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outliers  = dataset[(dataset['Parch']>max) |(dataset['Parch']<min) ].index","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.967487Z","iopub.execute_input":"2022-07-09T15:25:33.967836Z","iopub.status.idle":"2022-07-09T15:25:33.974302Z","shell.execute_reply.started":"2022-07-09T15:25:33.967804Z","shell.execute_reply":"2022-07-09T15:25:33.973304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\nCounter(Parch)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.975497Z","iopub.execute_input":"2022-07-09T15:25:33.976017Z","iopub.status.idle":"2022-07-09T15:25:33.989428Z","shell.execute_reply.started":"2022-07-09T15:25:33.975981Z","shell.execute_reply":"2022-07-09T15:25:33.988602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see , the number of points more than 2 are 15 ( 5 + 5 + 4 + 1) . So it would be easier  if we replace the values above 2 by 0 , as 0 is most frequent","metadata":{}},{"cell_type":"code","source":"greater_than_2 = dataset[(dataset['Parch']>2) ].index","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:33.990914Z","iopub.execute_input":"2022-07-09T15:25:33.991637Z","iopub.status.idle":"2022-07-09T15:25:33.999244Z","shell.execute_reply.started":"2022-07-09T15:25:33.991593Z","shell.execute_reply":"2022-07-09T15:25:33.998238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in greater_than_2:\n    dataset.loc[i , 'Parch'] = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.000326Z","iopub.execute_input":"2022-07-09T15:25:34.001302Z","iopub.status.idle":"2022-07-09T15:25:34.016358Z","shell.execute_reply.started":"2022-07-09T15:25:34.001264Z","shell.execute_reply":"2022-07-09T15:25:34.015240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"TicketNo","metadata":{}},{"cell_type":"markdown","source":"A Passenger's ticket number will not affect whether they survived or not. So we can drop the ticket number","metadata":{}},{"cell_type":"code","source":"dataset.drop(['Ticket'] , inplace = True , axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.017419Z","iopub.execute_input":"2022-07-09T15:25:34.018256Z","iopub.status.idle":"2022-07-09T15:25:34.029387Z","shell.execute_reply.started":"2022-07-09T15:25:34.018219Z","shell.execute_reply":"2022-07-09T15:25:34.028287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.032134Z","iopub.execute_input":"2022-07-09T15:25:34.033422Z","iopub.status.idle":"2022-07-09T15:25:34.052460Z","shell.execute_reply.started":"2022-07-09T15:25:34.033369Z","shell.execute_reply":"2022-07-09T15:25:34.051630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fare","metadata":{}},{"cell_type":"markdown","source":"Fare depends on Class . So they will be positively correlated . Hence , after data preprocesing is over . We can use \nPrincipal Component Analysis to reduce dimensions and reduce 1 collumn . ","metadata":{}},{"cell_type":"markdown","source":"Nevertheless , let us observe Fare data for any possible incorrect entries","metadata":{}},{"cell_type":"code","source":"Fare  = dataset['Fare']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.053799Z","iopub.execute_input":"2022-07-09T15:25:34.054771Z","iopub.status.idle":"2022-07-09T15:25:34.059606Z","shell.execute_reply.started":"2022-07-09T15:25:34.054716Z","shell.execute_reply":"2022-07-09T15:25:34.058433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Fare.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.061829Z","iopub.execute_input":"2022-07-09T15:25:34.062643Z","iopub.status.idle":"2022-07-09T15:25:34.075914Z","shell.execute_reply.started":"2022-07-09T15:25:34.062596Z","shell.execute_reply":"2022-07-09T15:25:34.075003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visual analyse","metadata":{}},{"cell_type":"code","source":"plt.hist(Fare)\nplt.title('Fare')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.076994Z","iopub.execute_input":"2022-07-09T15:25:34.077633Z","iopub.status.idle":"2022-07-09T15:25:34.274810Z","shell.execute_reply.started":"2022-07-09T15:25:34.077598Z","shell.execute_reply":"2022-07-09T15:25:34.273837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cabin","metadata":{}},{"cell_type":"code","source":"cabin = dataset['Cabin']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.276098Z","iopub.execute_input":"2022-07-09T15:25:34.276423Z","iopub.status.idle":"2022-07-09T15:25:34.280795Z","shell.execute_reply.started":"2022-07-09T15:25:34.276394Z","shell.execute_reply":"2022-07-09T15:25:34.279894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(cabin).info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.282160Z","iopub.execute_input":"2022-07-09T15:25:34.283092Z","iopub.status.idle":"2022-07-09T15:25:34.299179Z","shell.execute_reply.started":"2022-07-09T15:25:34.283047Z","shell.execute_reply":"2022-07-09T15:25:34.298019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, against a total of nearly 891 values , here only 204 are non null . \nThe remaining 687 values are blank . \nSo in this case , it is better to drop this entire collumn , rather than speculating nan values,and using it as Training data","metadata":{}},{"cell_type":"code","source":"dataset.drop(['Cabin'] , inplace = True , axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.300362Z","iopub.execute_input":"2022-07-09T15:25:34.301396Z","iopub.status.idle":"2022-07-09T15:25:34.306845Z","shell.execute_reply.started":"2022-07-09T15:25:34.301358Z","shell.execute_reply":"2022-07-09T15:25:34.306018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.308102Z","iopub.execute_input":"2022-07-09T15:25:34.308814Z","iopub.status.idle":"2022-07-09T15:25:34.330336Z","shell.execute_reply.started":"2022-07-09T15:25:34.308777Z","shell.execute_reply":"2022-07-09T15:25:34.329106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Embarked","metadata":{}},{"cell_type":"markdown","source":"The place where a person embarked is very useful here . \nEx : The person who embarked last , could have his cabin on the outer of cabins , while a person who embarked first could have his cabin inside the ship , causing them to have a time difference in reaching the boats","metadata":{}},{"cell_type":"code","source":"embarked = dataset['Embarked']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.331647Z","iopub.execute_input":"2022-07-09T15:25:34.332067Z","iopub.status.idle":"2022-07-09T15:25:34.336208Z","shell.execute_reply.started":"2022-07-09T15:25:34.332037Z","shell.execute_reply":"2022-07-09T15:25:34.335219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Counter(embarked)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.337751Z","iopub.execute_input":"2022-07-09T15:25:34.338174Z","iopub.status.idle":"2022-07-09T15:25:34.349157Z","shell.execute_reply.started":"2022-07-09T15:25:34.338143Z","shell.execute_reply":"2022-07-09T15:25:34.348069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since , there are only 2 np.nan values , we can replace then with 'S' . ","metadata":{}},{"cell_type":"code","source":"embarked = embarked.fillna('S')\ndataset['Embarked'] = embarked","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.350686Z","iopub.execute_input":"2022-07-09T15:25:34.351899Z","iopub.status.idle":"2022-07-09T15:25:34.359730Z","shell.execute_reply.started":"2022-07-09T15:25:34.351852Z","shell.execute_reply":"2022-07-09T15:25:34.358769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.361099Z","iopub.execute_input":"2022-07-09T15:25:34.362110Z","iopub.status.idle":"2022-07-09T15:25:34.376988Z","shell.execute_reply.started":"2022-07-09T15:25:34.362075Z","shell.execute_reply":"2022-07-09T15:25:34.375906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have now finished data preprocessing for the data.\nNow we should Encode the data , as Embarked is a string ","metadata":{}},{"cell_type":"markdown","source":"Encoding","metadata":{}},{"cell_type":"code","source":"X_train = dataset.iloc[: , 2:].values\ny_train = dataset.iloc[: , 1].values\nid = dataset.iloc[: , 0].values","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.378121Z","iopub.execute_input":"2022-07-09T15:25:34.378734Z","iopub.status.idle":"2022-07-09T15:25:34.391052Z","shell.execute_reply.started":"2022-07-09T15:25:34.378697Z","shell.execute_reply":"2022-07-09T15:25:34.390205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder\nct = ColumnTransformer(transformers=[('encoder', OneHotEncoder(), [-1])], remainder='passthrough')\nX_train = np.array(ct.fit_transform(X_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.392465Z","iopub.execute_input":"2022-07-09T15:25:34.392808Z","iopub.status.idle":"2022-07-09T15:25:34.928198Z","shell.execute_reply.started":"2022-07-09T15:25:34.392771Z","shell.execute_reply":"2022-07-09T15:25:34.927258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.929376Z","iopub.execute_input":"2022-07-09T15:25:34.929675Z","iopub.status.idle":"2022-07-09T15:25:34.936461Z","shell.execute_reply.started":"2022-07-09T15:25:34.929647Z","shell.execute_reply":"2022-07-09T15:25:34.935364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do Standard Scaling to increase accuracy","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\nX_train = sc.fit_transform(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.938113Z","iopub.execute_input":"2022-07-09T15:25:34.938571Z","iopub.status.idle":"2022-07-09T15:25:34.950809Z","shell.execute_reply.started":"2022-07-09T15:25:34.938529Z","shell.execute_reply":"2022-07-09T15:25:34.949868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do principal Component analysis to reduce dimensionality","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\npca = PCA(n_components = 2)\nX_train = pca.fit_transform(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:34.952720Z","iopub.execute_input":"2022-07-09T15:25:34.953614Z","iopub.status.idle":"2022-07-09T15:25:35.131064Z","shell.execute_reply.started":"2022-07-09T15:25:34.953567Z","shell.execute_reply":"2022-07-09T15:25:35.129678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = dataset.iloc[: , 1].values","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.132818Z","iopub.execute_input":"2022-07-09T15:25:35.133531Z","iopub.status.idle":"2022-07-09T15:25:35.139251Z","shell.execute_reply.started":"2022-07-09T15:25:35.133486Z","shell.execute_reply":"2022-07-09T15:25:35.138004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nclassifier = LogisticRegression()\nclassifier.fit(X_train , y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.141085Z","iopub.execute_input":"2022-07-09T15:25:35.141766Z","iopub.status.idle":"2022-07-09T15:25:35.165812Z","shell.execute_reply.started":"2022-07-09T15:25:35.141722Z","shell.execute_reply":"2022-07-09T15:25:35.164672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observe Training Data","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfrom mpl_toolkits.mplot3d import Axes3D\n\nfig = plt.figure(figsize=(4,4))\n\nax = fig.add_subplot(111, projection='3d')\nx = pd.DataFrame(X_train).iloc[: , 0].values\ny = pd.DataFrame(X_train).iloc[: , 1].values\nax.scatter(x , y , y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.167538Z","iopub.execute_input":"2022-07-09T15:25:35.168221Z","iopub.status.idle":"2022-07-09T15:25:35.363835Z","shell.execute_reply.started":"2022-07-09T15:25:35.168178Z","shell.execute_reply":"2022-07-09T15:25:35.362734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Import Test Data","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"../input/titanic/test.csv\")\nid_test = test.iloc[:, 0].values","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.365876Z","iopub.execute_input":"2022-07-09T15:25:35.366731Z","iopub.status.idle":"2022-07-09T15:25:35.381778Z","shell.execute_reply.started":"2022-07-09T15:25:35.366684Z","shell.execute_reply":"2022-07-09T15:25:35.380807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.383514Z","iopub.execute_input":"2022-07-09T15:25:35.383958Z","iopub.status.idle":"2022-07-09T15:25:35.399695Z","shell.execute_reply.started":"2022-07-09T15:25:35.383895Z","shell.execute_reply":"2022-07-09T15:25:35.398667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As done previously , we will drop passenger id  , Name , Cabin . ","metadata":{}},{"cell_type":"code","source":"id  = test.iloc[: , 0].values \ntest.drop(['Name' , 'Cabin' , 'Ticket'] , inplace = True , axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.401675Z","iopub.execute_input":"2022-07-09T15:25:35.403236Z","iopub.status.idle":"2022-07-09T15:25:35.409858Z","shell.execute_reply.started":"2022-07-09T15:25:35.403186Z","shell.execute_reply":"2022-07-09T15:25:35.408981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.411304Z","iopub.execute_input":"2022-07-09T15:25:35.412148Z","iopub.status.idle":"2022-07-09T15:25:35.428113Z","shell.execute_reply.started":"2022-07-09T15:25:35.412114Z","shell.execute_reply":"2022-07-09T15:25:35.427240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test Age","metadata":{}},{"cell_type":"code","source":"test_age = test['Age'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.429654Z","iopub.execute_input":"2022-07-09T15:25:35.430015Z","iopub.status.idle":"2022-07-09T15:25:35.435212Z","shell.execute_reply.started":"2022-07-09T15:25:35.429983Z","shell.execute_reply":"2022-07-09T15:25:35.434013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(test['Age']).describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.437060Z","iopub.execute_input":"2022-07-09T15:25:35.437592Z","iopub.status.idle":"2022-07-09T15:25:35.457763Z","shell.execute_reply.started":"2022-07-09T15:25:35.437547Z","shell.execute_reply":"2022-07-09T15:25:35.456628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(pd.DataFrame(test['Age']))\nplt.title('Test Age')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.459266Z","iopub.execute_input":"2022-07-09T15:25:35.459614Z","iopub.status.idle":"2022-07-09T15:25:35.659521Z","shell.execute_reply.started":"2022-07-09T15:25:35.459584Z","shell.execute_reply":"2022-07-09T15:25:35.658439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agedf = pd.DataFrame(test_age)\nnon_age_df = agedf.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.660890Z","iopub.execute_input":"2022-07-09T15:25:35.661839Z","iopub.status.idle":"2022-07-09T15:25:35.668378Z","shell.execute_reply.started":"2022-07-09T15:25:35.661801Z","shell.execute_reply":"2022-07-09T15:25:35.667338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(non_age_df.values , bins = 80)\nplt.title('Age')\nage_data = pd.DataFrame(test['Age'])\ntest['Age'] = age_data.replace(np.nan , median)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.670004Z","iopub.execute_input":"2022-07-09T15:25:35.670319Z","iopub.status.idle":"2022-07-09T15:25:35.997960Z","shell.execute_reply.started":"2022-07-09T15:25:35.670289Z","shell.execute_reply":"2022-07-09T15:25:35.996832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.boxplot(pd.DataFrame(test['Age']))\nplt.title('Test Age')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:35.999778Z","iopub.execute_input":"2022-07-09T15:25:36.000239Z","iopub.status.idle":"2022-07-09T15:25:36.175085Z","shell.execute_reply.started":"2022-07-09T15:25:36.000196Z","shell.execute_reply":"2022-07-09T15:25:36.173571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_age_df = test['Age']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:36.180023Z","iopub.execute_input":"2022-07-09T15:25:36.180451Z","iopub.status.idle":"2022-07-09T15:25:36.187747Z","shell.execute_reply.started":"2022-07-09T15:25:36.180403Z","shell.execute_reply":"2022-07-09T15:25:36.186196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have to remove outliers in test data again","metadata":{}},{"cell_type":"code","source":"Q1 = non_age_df.quantile(0.25)\nQ2 = non_age_df.quantile(0.50)\nQ3 = non_age_df.quantile(0.75)\nIQR = Q3 - Q1\nmin = Q1 - (1.5*IQR)\nmax = Q3 + (1.5*IQR)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:36.189925Z","iopub.execute_input":"2022-07-09T15:25:36.190638Z","iopub.status.idle":"2022-07-09T15:25:36.202725Z","shell.execute_reply.started":"2022-07-09T15:25:36.190590Z","shell.execute_reply":"2022-07-09T15:25:36.201585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"First Quartile is : \" , Q1)\nprint(\"Second Quartile is : \" , Q2)\nprint(\"Third Quartile is : \" , Q3)\nprint(\"IQR is : \" , IQR)\nprint(\"Minimum Value to not be an outlier is  : \" , min)\nprint(\"Maximum Value to not be an outlier is  :\" , max)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:25:36.204104Z","iopub.execute_input":"2022-07-09T15:25:36.205019Z","iopub.status.idle":"2022-07-09T15:25:36.215600Z","shell.execute_reply.started":"2022-07-09T15:25:36.204982Z","shell.execute_reply":"2022-07-09T15:25:36.214514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outliers =test[(test['Age']>max) | (test['Age']<min)].index\nprint(\"Number of Outliers : \" , len(outliers))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:06.286129Z","iopub.execute_input":"2022-07-09T15:26:06.286531Z","iopub.status.idle":"2022-07-09T15:26:06.294561Z","shell.execute_reply.started":"2022-07-09T15:26:06.286495Z","shell.execute_reply":"2022-07-09T15:26:06.293494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"median = non_age_df.quantile(0.5)\nfor i in list(outliers):\n    test.iloc[i , 4] = Q2","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.355387Z","iopub.execute_input":"2022-07-09T15:26:19.355812Z","iopub.status.idle":"2022-07-09T15:26:19.373815Z","shell.execute_reply.started":"2022-07-09T15:26:19.355775Z","shell.execute_reply":"2022-07-09T15:26:19.372591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.boxplot(test['Age'])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.388754Z","iopub.execute_input":"2022-07-09T15:26:19.389595Z","iopub.status.idle":"2022-07-09T15:26:19.556916Z","shell.execute_reply.started":"2022-07-09T15:26:19.389553Z","shell.execute_reply":"2022-07-09T15:26:19.555731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['Sex'] = test['Sex'].replace('male' , 1)\ntest['Sex'] = test['Sex'].replace('female' , 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.558774Z","iopub.execute_input":"2022-07-09T15:26:19.559233Z","iopub.status.idle":"2022-07-09T15:26:19.566320Z","shell.execute_reply.started":"2022-07-09T15:26:19.559197Z","shell.execute_reply":"2022-07-09T15:26:19.564867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fare","metadata":{}},{"cell_type":"code","source":"Fare = test['Fare']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.567762Z","iopub.execute_input":"2022-07-09T15:26:19.568156Z","iopub.status.idle":"2022-07-09T15:26:19.576635Z","shell.execute_reply.started":"2022-07-09T15:26:19.568119Z","shell.execute_reply":"2022-07-09T15:26:19.575770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"median = pd.DataFrame(Fare).quantile(0.5)\ntest['Fare'] = Fare.fillna(median[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.579610Z","iopub.execute_input":"2022-07-09T15:26:19.580719Z","iopub.status.idle":"2022-07-09T15:26:19.590207Z","shell.execute_reply.started":"2022-07-09T15:26:19.580663Z","shell.execute_reply":"2022-07-09T15:26:19.588976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.592867Z","iopub.execute_input":"2022-07-09T15:26:19.593592Z","iopub.status.idle":"2022-07-09T15:26:19.608417Z","shell.execute_reply.started":"2022-07-09T15:26:19.593552Z","shell.execute_reply":"2022-07-09T15:26:19.607227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Scale and do pca","metadata":{}},{"cell_type":"code","source":"X_test = test.iloc[: , 1:].values ","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.609609Z","iopub.execute_input":"2022-07-09T15:26:19.609952Z","iopub.status.idle":"2022-07-09T15:26:19.616882Z","shell.execute_reply.started":"2022-07-09T15:26:19.609898Z","shell.execute_reply":"2022-07-09T15:26:19.615422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Encoding","metadata":{}},{"cell_type":"markdown","source":"Note that for encoding , standard scalar and Principal Component analysis , we must use the same class object used for training data  set","metadata":{}},{"cell_type":"code","source":"X_test = np.array(ct.fit_transform(X_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.618999Z","iopub.execute_input":"2022-07-09T15:26:19.619487Z","iopub.status.idle":"2022-07-09T15:26:19.630722Z","shell.execute_reply.started":"2022-07-09T15:26:19.619438Z","shell.execute_reply":"2022-07-09T15:26:19.629512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = sc.fit_transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.632207Z","iopub.execute_input":"2022-07-09T15:26:19.633280Z","iopub.status.idle":"2022-07-09T15:26:19.641906Z","shell.execute_reply.started":"2022-07-09T15:26:19.633234Z","shell.execute_reply":"2022-07-09T15:26:19.640829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(X_test).info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.643129Z","iopub.execute_input":"2022-07-09T15:26:19.644211Z","iopub.status.idle":"2022-07-09T15:26:19.662480Z","shell.execute_reply.started":"2022-07-09T15:26:19.644170Z","shell.execute_reply":"2022-07-09T15:26:19.661147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pca.fit_transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.666798Z","iopub.execute_input":"2022-07-09T15:26:19.667281Z","iopub.status.idle":"2022-07-09T15:26:19.679091Z","shell.execute_reply.started":"2022-07-09T15:26:19.667209Z","shell.execute_reply":"2022-07-09T15:26:19.678028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Prediction","metadata":{}},{"cell_type":"code","source":"X_test","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.680548Z","iopub.execute_input":"2022-07-09T15:26:19.681097Z","iopub.status.idle":"2022-07-09T15:26:19.705724Z","shell.execute_reply.started":"2022-07-09T15:26:19.681044Z","shell.execute_reply":"2022-07-09T15:26:19.704782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = classifier.predict(X_test)\nfp = open('submission.csv' , 'w')\nimport csv\nwriter = csv.writer(fp)\nheader = ['PassengerId' , 'Survived']\nwriter.writerow(header)\nfor i in range(418):\n    data = [id[i] , result[i]]\n    writer.writerow(data)\nfp.close()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T15:26:19.707235Z","iopub.execute_input":"2022-07-09T15:26:19.707594Z","iopub.status.idle":"2022-07-09T15:26:19.716822Z","shell.execute_reply.started":"2022-07-09T15:26:19.707562Z","shell.execute_reply":"2022-07-09T15:26:19.715527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create File","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}