{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\nsns.set_style(\"dark\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T16:30:11.240077Z","iopub.execute_input":"2022-07-09T16:30:11.241564Z","iopub.status.idle":"2022-07-09T16:30:11.253594Z","shell.execute_reply.started":"2022-07-09T16:30:11.241497Z","shell.execute_reply":"2022-07-09T16:30:11.252321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Load Dataset","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/titanic/train.csv\")\ntest = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.296793Z","iopub.execute_input":"2022-07-09T16:30:11.299632Z","iopub.status.idle":"2022-07-09T16:30:11.331999Z","shell.execute_reply.started":"2022-07-09T16:30:11.299560Z","shell.execute_reply":"2022-07-09T16:30:11.330601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train set shape: {train.shape}\")\nprint(f\"Test set shape: {test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.334961Z","iopub.execute_input":"2022-07-09T16:30:11.335544Z","iopub.status.idle":"2022-07-09T16:30:11.344241Z","shell.execute_reply.started":"2022-07-09T16:30:11.335494Z","shell.execute_reply":"2022-07-09T16:30:11.342347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Merge train and test data","metadata":{}},{"cell_type":"code","source":"df = pd.concat([train,test])\nprint(f\"Combined dataset shape: {df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.365483Z","iopub.execute_input":"2022-07-09T16:30:11.366389Z","iopub.status.idle":"2022-07-09T16:30:11.380361Z","shell.execute_reply.started":"2022-07-09T16:30:11.366333Z","shell.execute_reply":"2022-07-09T16:30:11.379343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Overview","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.401001Z","iopub.execute_input":"2022-07-09T16:30:11.401499Z","iopub.status.idle":"2022-07-09T16:30:11.426965Z","shell.execute_reply.started":"2022-07-09T16:30:11.401461Z","shell.execute_reply":"2022-07-09T16:30:11.425725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* column survived is the target column hence after merging test set it will have missing data.\n* column age seems to have some missing values which can be handled\n* column Fare also has 1 missing value which is easy to fix\n* Cabin has most missing values thus dropping it would be the best option\n* PassengerId is just an identification integer thus can be dropped","metadata":{}},{"cell_type":"code","source":"df.dtypes.value_counts().plot(kind=\"bar\", title=\"Columns By Data Types\")\nplt.xlabel(\"Data Types\")\nplt.ylabel(\"Column Count\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.432267Z","iopub.execute_input":"2022-07-09T16:30:11.432998Z","iopub.status.idle":"2022-07-09T16:30:11.651665Z","shell.execute_reply.started":"2022-07-09T16:30:11.432944Z","shell.execute_reply":"2022-07-09T16:30:11.650442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are more categorical variables than continuous ones","metadata":{}},{"cell_type":"markdown","source":"#### Explore categorical variables","metadata":{}},{"cell_type":"code","source":"df.select_dtypes(object).head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.654509Z","iopub.execute_input":"2022-07-09T16:30:11.654985Z","iopub.status.idle":"2022-07-09T16:30:11.673682Z","shell.execute_reply.started":"2022-07-09T16:30:11.654938Z","shell.execute_reply":"2022-07-09T16:30:11.672188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Exploring continuous variables","metadata":{}},{"cell_type":"code","source":"df.select_dtypes(\"number\").head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.675836Z","iopub.execute_input":"2022-07-09T16:30:11.676461Z","iopub.status.idle":"2022-07-09T16:30:11.695112Z","shell.execute_reply.started":"2022-07-09T16:30:11.676413Z","shell.execute_reply":"2022-07-09T16:30:11.693894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Cleaning\n\n1. Drop PassengerId\n2. Drop Cabin\n3. Fill missing values for age\n4. Fill missing values for Fare","metadata":{}},{"cell_type":"code","source":"# drop cabin and passengerId columns\ndf.drop([\"PassengerId\",\"Cabin\"],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.697479Z","iopub.execute_input":"2022-07-09T16:30:11.698169Z","iopub.status.idle":"2022-07-09T16:30:11.704797Z","shell.execute_reply.started":"2022-07-09T16:30:11.698115Z","shell.execute_reply":"2022-07-09T16:30:11.703911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Handling missing values for Age","metadata":{}},{"cell_type":"code","source":"fig,ax = plt.subplots(figsize=(7,7))\nage = df['Age'].dropna()\nsns.histplot(data=age,ax=ax).set_title(\"Age Distribution (Before Cleaning)\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:11.705866Z","iopub.execute_input":"2022-07-09T16:30:11.706164Z","iopub.status.idle":"2022-07-09T16:30:12.045616Z","shell.execute_reply.started":"2022-07-09T16:30:11.706119Z","shell.execute_reply":"2022-07-09T16:30:12.044373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There were more people on the ship whose age was between 20 to 40 years.","metadata":{}},{"cell_type":"code","source":"# Fill missing values with the mode of age\ndf['Age'] = df['Age'].fillna(df['Age'].median())","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:12.047134Z","iopub.execute_input":"2022-07-09T16:30:12.048309Z","iopub.status.idle":"2022-07-09T16:30:12.055652Z","shell.execute_reply.started":"2022-07-09T16:30:12.048273Z","shell.execute_reply":"2022-07-09T16:30:12.054337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax = plt.subplots(figsize=(7,7))\nage = df['Age'].dropna()\nsns.histplot(data=age,ax=ax).set_title(\"Age Distribution (After Cleaning)\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:12.057462Z","iopub.execute_input":"2022-07-09T16:30:12.057932Z","iopub.status.idle":"2022-07-09T16:30:12.402789Z","shell.execute_reply.started":"2022-07-09T16:30:12.057888Z","shell.execute_reply":"2022-07-09T16:30:12.401486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Handling missing values for Fare","metadata":{}},{"cell_type":"code","source":"fig,ax = plt.subplots(figsize=(7,7))\nage = df['Fare'].dropna()\nsns.histplot(data=age,ax=ax).set_title(\" Fare Distribution (Before Cleaning)\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:12.404769Z","iopub.execute_input":"2022-07-09T16:30:12.405162Z","iopub.status.idle":"2022-07-09T16:30:12.881256Z","shell.execute_reply.started":"2022-07-09T16:30:12.405115Z","shell.execute_reply":"2022-07-09T16:30:12.879917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill Fare values by its mode\ndf['Fare'] = df['Fare'].fillna(df['Fare'].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:12.885710Z","iopub.execute_input":"2022-07-09T16:30:12.886075Z","iopub.status.idle":"2022-07-09T16:30:12.892290Z","shell.execute_reply.started":"2022-07-09T16:30:12.886043Z","shell.execute_reply":"2022-07-09T16:30:12.891545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Handling missing values for Emabarked","metadata":{}},{"cell_type":"code","source":"fig,ax = plt.subplots(figsize=(7,7))\nage = df['Embarked'].dropna()\nsns.histplot(data=age,ax=ax).set_title(\"Embarked Distribution\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:12.893704Z","iopub.execute_input":"2022-07-09T16:30:12.894063Z","iopub.status.idle":"2022-07-09T16:30:13.098280Z","shell.execute_reply.started":"2022-07-09T16:30:12.894030Z","shell.execute_reply":"2022-07-09T16:30:13.097320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* S = Southhamption\n* C = Cherbourg\n* Q = Queenstown\n\nMost people embarked from Southhampton","metadata":{}},{"cell_type":"code","source":"df['Embarked'] = df['Embarked'].fillna(df['Embarked'].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:13.102019Z","iopub.execute_input":"2022-07-09T16:30:13.103161Z","iopub.status.idle":"2022-07-09T16:30:13.110011Z","shell.execute_reply.started":"2022-07-09T16:30:13.103108Z","shell.execute_reply":"2022-07-09T16:30:13.108837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:13.111591Z","iopub.execute_input":"2022-07-09T16:30:13.112249Z","iopub.status.idle":"2022-07-09T16:30:13.123332Z","shell.execute_reply.started":"2022-07-09T16:30:13.112216Z","shell.execute_reply":"2022-07-09T16:30:13.122353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:13.124850Z","iopub.execute_input":"2022-07-09T16:30:13.125979Z","iopub.status.idle":"2022-07-09T16:30:13.142273Z","shell.execute_reply.started":"2022-07-09T16:30:13.125938Z","shell.execute_reply":"2022-07-09T16:30:13.141335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Univariate analysis","metadata":{}},{"cell_type":"code","source":"def show_univariate_chart(df, variable,title):\n    fig,ax = plt.subplots(figsize=(7,7))\n    age = df[variable].dropna()\n    age.value_counts().plot(kind=\"bar\",title=f\"{title} Distribution\")\n    plt.xlabel(title)\n    plt.ylabel(\"Count\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:13.143523Z","iopub.execute_input":"2022-07-09T16:30:13.144607Z","iopub.status.idle":"2022-07-09T16:30:13.151481Z","shell.execute_reply.started":"2022-07-09T16:30:13.144569Z","shell.execute_reply":"2022-07-09T16:30:13.150180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_univariate_chart(df,\"Pclass\",\"Ticket Class\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:13.153129Z","iopub.execute_input":"2022-07-09T16:30:13.153999Z","iopub.status.idle":"2022-07-09T16:30:13.357086Z","shell.execute_reply.started":"2022-07-09T16:30:13.153948Z","shell.execute_reply":"2022-07-09T16:30:13.355904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 3 = Lower Class\n* 2 = Middle Class\n* 1 = Upper Class\n\nThere were more lower class people on the ship","metadata":{}},{"cell_type":"code","source":"show_univariate_chart(df,\"Sex\",\"Sex\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:13.358991Z","iopub.execute_input":"2022-07-09T16:30:13.359744Z","iopub.status.idle":"2022-07-09T16:30:13.555543Z","shell.execute_reply.started":"2022-07-09T16:30:13.359697Z","shell.execute_reply":"2022-07-09T16:30:13.554195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There were more males than females on the ship","metadata":{}},{"cell_type":"code","source":"show_univariate_chart(df,\"SibSp\",\"Siblings and Spouse\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:13.557706Z","iopub.execute_input":"2022-07-09T16:30:13.558516Z","iopub.status.idle":"2022-07-09T16:30:13.775496Z","shell.execute_reply.started":"2022-07-09T16:30:13.558464Z","shell.execute_reply":"2022-07-09T16:30:13.774509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_univariate_chart(df,\"Parch\",\"Parents and Children\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:13.776978Z","iopub.execute_input":"2022-07-09T16:30:13.777602Z","iopub.status.idle":"2022-07-09T16:30:14.004232Z","shell.execute_reply.started":"2022-07-09T16:30:13.777569Z","shell.execute_reply":"2022-07-09T16:30:14.003348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Corelations","metadata":{}},{"cell_type":"code","source":"sns.heatmap(df.dropna().corr(), cmap=\"Oranges\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:14.005369Z","iopub.execute_input":"2022-07-09T16:30:14.006259Z","iopub.status.idle":"2022-07-09T16:30:14.301267Z","shell.execute_reply.started":"2022-07-09T16:30:14.006225Z","shell.execute_reply":"2022-07-09T16:30:14.300417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(7,7))\nsns.histplot(data=df.dropna(),x=\"Age\",hue=\"Survived\",multiple=\"stack\",ax=ax).set_title(\"Age to Survival Ratio\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:14.302571Z","iopub.execute_input":"2022-07-09T16:30:14.303649Z","iopub.status.idle":"2022-07-09T16:30:14.737201Z","shell.execute_reply.started":"2022-07-09T16:30:14.303615Z","shell.execute_reply":"2022-07-09T16:30:14.736032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(7,7))\nsns.histplot(data=df.dropna(),x=\"Sex\",hue=\"Survived\",multiple=\"stack\",ax=ax).set_title(\"Gender to Survival Ratio\")\nplt.xlabel(\"Gender\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:14.738676Z","iopub.execute_input":"2022-07-09T16:30:14.739118Z","iopub.status.idle":"2022-07-09T16:30:14.984963Z","shell.execute_reply.started":"2022-07-09T16:30:14.739084Z","shell.execute_reply":"2022-07-09T16:30:14.983937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Females were prioritized during the evacuation","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(7,7))\nsns.histplot(data=df.dropna(),x=\"Pclass\",hue=\"Survived\",multiple=\"stack\",ax=ax).set_title(\"Passenger Class to Survival Ratio\")\nplt.xlabel(\"Passenger Class\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:14.990671Z","iopub.execute_input":"2022-07-09T16:30:14.991027Z","iopub.status.idle":"2022-07-09T16:30:15.353238Z","shell.execute_reply.started":"2022-07-09T16:30:14.990998Z","shell.execute_reply":"2022-07-09T16:30:15.352152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First class passengers were given more priority during evacuation","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(7,7))\nsns.histplot(data=df.dropna(),x=\"Embarked\",hue=\"Survived\",multiple=\"stack\",ax=ax).set_title(\"Embarked to Survival Ratio\")\nplt.xlabel(\"Embarked From\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:15.354650Z","iopub.execute_input":"2022-07-09T16:30:15.354982Z","iopub.status.idle":"2022-07-09T16:30:15.608568Z","shell.execute_reply.started":"2022-07-09T16:30:15.354952Z","shell.execute_reply":"2022-07-09T16:30:15.607377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"People who embarked from Southhampton had more survival ratio than people who emabarked from somewhere else","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(7,7))\nsns.histplot(data=df.dropna(),x=\"SibSp\",hue=\"Survived\",multiple=\"stack\",ax=ax).set_title(\"Sibling/Spouse to Survival Ratio\")\nplt.xlabel(\"Siblings / Spouse\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:15.610180Z","iopub.execute_input":"2022-07-09T16:30:15.610507Z","iopub.status.idle":"2022-07-09T16:30:16.093181Z","shell.execute_reply.started":"2022-07-09T16:30:15.610477Z","shell.execute_reply":"2022-07-09T16:30:16.091781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(7,7))\nsns.histplot(data=df.dropna(),x=\"Parch\",hue=\"Survived\",multiple=\"stack\",ax=ax).set_title(\"Parent/Children Count to Survival Ratio\")\nplt.xlabel(\"Parent/Children Count\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.094614Z","iopub.execute_input":"2022-07-09T16:30:16.094934Z","iopub.status.idle":"2022-07-09T16:30:16.446878Z","shell.execute_reply.started":"2022-07-09T16:30:16.094905Z","shell.execute_reply":"2022-07-09T16:30:16.445725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check if row with parch value 3 is an outlier or not?","metadata":{}},{"cell_type":"markdown","source":"#### Passenger Initials Importance","metadata":{}},{"cell_type":"code","source":"def get_initial(name):\n    return name.split(\",\")[1].split(\". \")[0]\n\ndf['Initial'] = df['Name'].apply(lambda x: get_initial(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.448363Z","iopub.execute_input":"2022-07-09T16:30:16.448792Z","iopub.status.idle":"2022-07-09T16:30:16.458446Z","shell.execute_reply.started":"2022-07-09T16:30:16.448759Z","shell.execute_reply":"2022-07-09T16:30:16.457063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(10,10))\nsns.histplot(data=df.dropna(),y=\"Initial\",hue=\"Survived\",multiple=\"stack\",ax=ax).set_title(\"Initial to Survival Ratio\")\nplt.xlabel(\"Initials\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.459972Z","iopub.execute_input":"2022-07-09T16:30:16.460377Z","iopub.status.idle":"2022-07-09T16:30:16.876014Z","shell.execute_reply.started":"2022-07-09T16:30:16.460344Z","shell.execute_reply":"2022-07-09T16:30:16.875129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Regression\n---\n\n1. Feature Selection\n2. Convert categorical datapoints to numericals\n3. Split test and training data\n4. Test on different models\n5. Select the one with best accuracy","metadata":{}},{"cell_type":"markdown","source":"#### Handle Categorical Values","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.877525Z","iopub.execute_input":"2022-07-09T16:30:16.878469Z","iopub.status.idle":"2022-07-09T16:30:16.883575Z","shell.execute_reply.started":"2022-07-09T16:30:16.878419Z","shell.execute_reply":"2022-07-09T16:30:16.882312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sex_le = LabelEncoder()\nsex_le.fit(df.Sex)\nsex_transformed = sex_le.transform(df.Sex)\ndf['Sex'] = sex_transformed","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.885372Z","iopub.execute_input":"2022-07-09T16:30:16.885929Z","iopub.status.idle":"2022-07-09T16:30:16.897310Z","shell.execute_reply.started":"2022-07-09T16:30:16.885883Z","shell.execute_reply":"2022-07-09T16:30:16.896205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embarked_le = LabelEncoder()\nembarked_le.fit(df.Embarked)\nembarked_transformed = embarked_le.transform(df.Embarked)\ndf['Embarked'] = embarked_transformed","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.898803Z","iopub.execute_input":"2022-07-09T16:30:16.899270Z","iopub.status.idle":"2022-07-09T16:30:16.910099Z","shell.execute_reply.started":"2022-07-09T16:30:16.899227Z","shell.execute_reply":"2022-07-09T16:30:16.909206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_le = LabelEncoder()\ninitial_le.fit(df['Initial'])\ninitial_transformed = initial_le.transform(df['Initial'])\ndf['Initial'] = initial_transformed","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.911654Z","iopub.execute_input":"2022-07-09T16:30:16.911959Z","iopub.status.idle":"2022-07-09T16:30:16.923281Z","shell.execute_reply.started":"2022-07-09T16:30:16.911930Z","shell.execute_reply":"2022-07-09T16:30:16.922208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age'] = df['Age'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.924603Z","iopub.execute_input":"2022-07-09T16:30:16.925412Z","iopub.status.idle":"2022-07-09T16:30:16.937665Z","shell.execute_reply.started":"2022-07-09T16:30:16.925381Z","shell.execute_reply":"2022-07-09T16:30:16.936461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Feature Selection","metadata":{}},{"cell_type":"code","source":"irrelevant_features = ['Ticket','Fare','Name']\nmodel_df = df.drop(irrelevant_features,axis=1)\nmodel_df.dropna(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.940115Z","iopub.execute_input":"2022-07-09T16:30:16.940787Z","iopub.status.idle":"2022-07-09T16:30:16.953222Z","shell.execute_reply.started":"2022-07-09T16:30:16.940754Z","shell.execute_reply":"2022-07-09T16:30:16.952082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare Train Test Data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.954931Z","iopub.execute_input":"2022-07-09T16:30:16.955341Z","iopub.status.idle":"2022-07-09T16:30:16.961311Z","shell.execute_reply.started":"2022-07-09T16:30:16.955304Z","shell.execute_reply":"2022-07-09T16:30:16.960206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = model_df['Survived']\nx = model_df.drop(['Survived'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.962629Z","iopub.execute_input":"2022-07-09T16:30:16.962942Z","iopub.status.idle":"2022-07-09T16:30:16.975454Z","shell.execute_reply.started":"2022-07-09T16:30:16.962913Z","shell.execute_reply":"2022-07-09T16:30:16.974498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train,y_test = train_test_split(x, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.976978Z","iopub.execute_input":"2022-07-09T16:30:16.977571Z","iopub.status.idle":"2022-07-09T16:30:16.987084Z","shell.execute_reply.started":"2022-07-09T16:30:16.977537Z","shell.execute_reply":"2022-07-09T16:30:16.985859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Build Classification Model","metadata":{}},{"cell_type":"code","source":"from sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.989123Z","iopub.execute_input":"2022-07-09T16:30:16.990230Z","iopub.status.idle":"2022-07-09T16:30:16.997891Z","shell.execute_reply.started":"2022-07-09T16:30:16.990182Z","shell.execute_reply":"2022-07-09T16:30:16.997095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlog_reg = LogisticRegression()\nlog_reg.fit(x_train, y_train)\ny_pred = log_reg.predict(x_test)\nprint(metrics.accuracy_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:16.999284Z","iopub.execute_input":"2022-07-09T16:30:17.000377Z","iopub.status.idle":"2022-07-09T16:30:17.042166Z","shell.execute_reply.started":"2022-07-09T16:30:17.000333Z","shell.execute_reply":"2022-07-09T16:30:17.040805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nrandom_forest = RandomForestClassifier()\nrandom_forest.fit(x_train, y_train)\nfr_pred = random_forest.predict(x_test)\nprint(metrics.accuracy_score(y_test, fr_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.044083Z","iopub.execute_input":"2022-07-09T16:30:17.044553Z","iopub.status.idle":"2022-07-09T16:30:17.284214Z","shell.execute_reply.started":"2022-07-09T16:30:17.044510Z","shell.execute_reply":"2022-07-09T16:30:17.283369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Support Vector Machines","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\nsvc = SVC()\nsvc.fit(x_train, y_train)\nsvc_pred = svc.predict(x_test)\nprint(metrics.accuracy_score(y_test, svc_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.285727Z","iopub.execute_input":"2022-07-09T16:30:17.286498Z","iopub.status.idle":"2022-07-09T16:30:17.326074Z","shell.execute_reply.started":"2022-07-09T16:30:17.286446Z","shell.execute_reply":"2022-07-09T16:30:17.325217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Decision Tree Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\ndc = DecisionTreeClassifier()\ndc.fit(x_train, y_train)\ndc_pred = dc.predict(x_test)\nprint(metrics.accuracy_score(y_test, dc_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.327904Z","iopub.execute_input":"2022-07-09T16:30:17.328714Z","iopub.status.idle":"2022-07-09T16:30:17.341487Z","shell.execute_reply.started":"2022-07-09T16:30:17.328668Z","shell.execute_reply":"2022-07-09T16:30:17.340462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing on actual test set","metadata":{}},{"cell_type":"code","source":"test_df = df[df['Survived'].isnull()]\ntest_df = test_df.drop(['Survived'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.342743Z","iopub.execute_input":"2022-07-09T16:30:17.343425Z","iopub.status.idle":"2022-07-09T16:30:17.351643Z","shell.execute_reply.started":"2022-07-09T16:30:17.343372Z","shell.execute_reply":"2022-07-09T16:30:17.350623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.drop(irrelevant_features,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.353031Z","iopub.execute_input":"2022-07-09T16:30:17.353961Z","iopub.status.idle":"2022-07-09T16:30:17.361892Z","shell.execute_reply.started":"2022-07-09T16:30:17.353917Z","shell.execute_reply":"2022-07-09T16:30:17.360924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set_pred = log_reg.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.363388Z","iopub.execute_input":"2022-07-09T16:30:17.364278Z","iopub.status.idle":"2022-07-09T16:30:17.376556Z","shell.execute_reply.started":"2022-07-09T16:30:17.364245Z","shell.execute_reply":"2022-07-09T16:30:17.375340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Create Submission Dataframe","metadata":{}},{"cell_type":"code","source":"passenger_ids = list(test['PassengerId'])\nsubmission_df = pd.DataFrame({\"PassengerId\":passenger_ids, \"Survived\":test_set_pred})\nsubmission_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.377830Z","iopub.execute_input":"2022-07-09T16:30:17.378448Z","iopub.status.idle":"2022-07-09T16:30:17.393265Z","shell.execute_reply.started":"2022-07-09T16:30:17.378367Z","shell.execute_reply":"2022-07-09T16:30:17.392218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['Survived'] = submission_df['Survived'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.394646Z","iopub.execute_input":"2022-07-09T16:30:17.395732Z","iopub.status.idle":"2022-07-09T16:30:17.405526Z","shell.execute_reply.started":"2022-07-09T16:30:17.395696Z","shell.execute_reply":"2022-07-09T16:30:17.404093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission_int.csv\",index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:30:17.407128Z","iopub.execute_input":"2022-07-09T16:30:17.407536Z","iopub.status.idle":"2022-07-09T16:30:17.418727Z","shell.execute_reply.started":"2022-07-09T16:30:17.407505Z","shell.execute_reply":"2022-07-09T16:30:17.417672Z"},"trusted":true},"execution_count":null,"outputs":[]}]}