{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries:","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom sklearn.ensemble import RandomForestClassifier\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T16:59:00.782459Z","iopub.execute_input":"2022-08-08T16:59:00.783206Z","iopub.status.idle":"2022-08-08T16:59:00.793004Z","shell.execute_reply.started":"2022-08-08T16:59:00.783159Z","shell.execute_reply":"2022-08-08T16:59:00.790685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing the dataset:","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/titanic/train.csv\")\ntest = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:01:13.045494Z","iopub.execute_input":"2022-08-08T17:01:13.045936Z","iopub.status.idle":"2022-08-08T17:01:13.077263Z","shell.execute_reply.started":"2022-08-08T17:01:13.045902Z","shell.execute_reply":"2022-08-08T17:01:13.076297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Exploring the dataset:","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:01:47.830013Z","iopub.execute_input":"2022-08-08T17:01:47.830498Z","iopub.status.idle":"2022-08-08T17:01:47.858934Z","shell.execute_reply.started":"2022-08-08T17:01:47.830460Z","shell.execute_reply":"2022-08-08T17:01:47.857921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:02:02.103963Z","iopub.execute_input":"2022-08-08T17:02:02.105315Z","iopub.status.idle":"2022-08-08T17:02:02.112871Z","shell.execute_reply.started":"2022-08-08T17:02:02.105269Z","shell.execute_reply":"2022-08-08T17:02:02.111929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:02:09.665977Z","iopub.execute_input":"2022-08-08T17:02:09.666501Z","iopub.status.idle":"2022-08-08T17:02:09.714188Z","shell.execute_reply.started":"2022-08-08T17:02:09.666457Z","shell.execute_reply":"2022-08-08T17:02:09.713239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:02:28.589742Z","iopub.execute_input":"2022-08-08T17:02:28.590135Z","iopub.status.idle":"2022-08-08T17:02:28.609824Z","shell.execute_reply.started":"2022-08-08T17:02:28.590105Z","shell.execute_reply":"2022-08-08T17:02:28.608674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Exploratory Data Analysis:","metadata":{}},{"cell_type":"code","source":"sns.set_theme(style = 'whitegrid')\nsns.countplot(train['Survived'])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:02:56.646115Z","iopub.execute_input":"2022-08-08T17:02:56.646526Z","iopub.status.idle":"2022-08-08T17:02:56.867452Z","shell.execute_reply.started":"2022-08-08T17:02:56.646494Z","shell.execute_reply":"2022-08-08T17:02:56.866402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train['Pclass'])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:03:14.596911Z","iopub.execute_input":"2022-08-08T17:03:14.598093Z","iopub.status.idle":"2022-08-08T17:03:14.788291Z","shell.execute_reply.started":"2022-08-08T17:03:14.598052Z","shell.execute_reply":"2022-08-08T17:03:14.786946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:03:26.646788Z","iopub.execute_input":"2022-08-08T17:03:26.647414Z","iopub.status.idle":"2022-08-08T17:03:26.834043Z","shell.execute_reply.started":"2022-08-08T17:03:26.647378Z","shell.execute_reply":"2022-08-08T17:03:26.832983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train['SibSp'])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:03:36.801801Z","iopub.execute_input":"2022-08-08T17:03:36.802246Z","iopub.status.idle":"2022-08-08T17:03:36.972274Z","shell.execute_reply.started":"2022-08-08T17:03:36.802186Z","shell.execute_reply":"2022-08-08T17:03:36.970570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train['Parch'])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:03:52.583024Z","iopub.execute_input":"2022-08-08T17:03:52.584378Z","iopub.status.idle":"2022-08-08T17:03:52.817709Z","shell.execute_reply.started":"2022-08-08T17:03:52.584320Z","shell.execute_reply":"2022-08-08T17:03:52.816444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train['Embarked'])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:05:30.102298Z","iopub.execute_input":"2022-08-08T17:05:30.102696Z","iopub.status.idle":"2022-08-08T17:05:30.299731Z","shell.execute_reply.started":"2022-08-08T17:05:30.102664Z","shell.execute_reply":"2022-08-08T17:05:30.298673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering:","metadata":{}},{"cell_type":"code","source":"# Dropping useless columns such as PassengerId, Name & Ticket\n# Dropping the Cabin Column because it contains very high number of null values\n# Filling the null values in the Age column with it's mean\n# Filling the null values in the Embarked column with it's mode\n# Changing the values of Age column where age is less than 1\n# Based on counts, I have combined 'Parch' & 'SibSp'\n\ntrain = train.drop(['PassengerId','Cabin','Name','Ticket'], axis = 1)\ntrain['Age'] = train['Age'].fillna(train['Age'].mean())\ntrain['Embarked'] = train['Embarked'].fillna(train['Embarked'].mode()[0])\ntrain['Age'] = np.where(train['Age'] < 1, 1, train['Age'])\ntrain['Parch'] = np.where(train['Parch'] > 2, '3+', train['Parch'])\ntrain['SibSp'] = np.where(train['SibSp'] > 1, '2+', train['SibSp'])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:15:50.369511Z","iopub.execute_input":"2022-08-08T17:15:50.369931Z","iopub.status.idle":"2022-08-08T17:15:50.385176Z","shell.execute_reply.started":"2022-08-08T17:15:50.369901Z","shell.execute_reply":"2022-08-08T17:15:50.384086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replaing Categorical variables to numbers\ntrain.Sex.replace('male', 0, inplace = True)\ntrain.Sex.replace('female', 1, inplace = True)\n\ntrain.Embarked.replace('S', 0, inplace = True)\ntrain.Embarked.replace('C', 1, inplace = True)\ntrain.Embarked.replace('Q', 2, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:16:22.706357Z","iopub.execute_input":"2022-08-08T17:16:22.706734Z","iopub.status.idle":"2022-08-08T17:16:22.717190Z","shell.execute_reply.started":"2022-08-08T17:16:22.706703Z","shell.execute_reply":"2022-08-08T17:16:22.716275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Applying the same transformation to test dataset:","metadata":{}},{"cell_type":"code","source":"test = test.drop(['PassengerId','Cabin','Name','Ticket'], axis = 1)\ntest['Age'] = test['Age'].fillna(test['Age'].mean())\n\n#Here one value in Fare was missing.\ntest['Fare'] = test['Fare'].fillna(test['Fare'].mean())\n\ntest['Embarked'] = test['Embarked'].fillna(test['Embarked'].mode()[0])\ntest['Age'] = np.where(test['Age'] < 1, 1, test['Age'])\ntest['Parch'] = np.where(test['Parch'] > 2, '3+', test['Parch'])\ntest['SibSp'] = np.where(test['SibSp'] > 1, '2+', test['SibSp'])\n\ntest.Sex.replace('male', 0, inplace = True)\ntest.Sex.replace('female', 1, inplace = True)\n\ntest.Embarked.replace('S', 0, inplace = True)\ntest.Embarked.replace('C', 1, inplace = True)\ntest.Embarked.replace('Q', 2, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:17:28.835492Z","iopub.execute_input":"2022-08-08T17:17:28.836534Z","iopub.status.idle":"2022-08-08T17:17:28.856012Z","shell.execute_reply.started":"2022-08-08T17:17:28.836493Z","shell.execute_reply":"2022-08-08T17:17:28.854991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest Classification:","metadata":{}},{"cell_type":"code","source":"y = train[\"Survived\"]\n\nfeatures = [\"Pclass\", \"Sex\", \"SibSp\", \"Parch\", \"Age\", \"Fare\", \"Embarked\"]\nx = pd.get_dummies(train[features])\nx_test = pd.get_dummies(test[features])\n\nmodel = RandomForestClassifier(n_estimators=100, max_depth=5, random_state=1)\nmodel.fit(x, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:19:45.138389Z","iopub.execute_input":"2022-08-08T17:19:45.138787Z","iopub.status.idle":"2022-08-08T17:19:45.341443Z","shell.execute_reply.started":"2022-08-08T17:19:45.138757Z","shell.execute_reply":"2022-08-08T17:19:45.340252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(x_test)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-08T17:20:10.037999Z","iopub.execute_input":"2022-08-08T17:20:10.038464Z","iopub.status.idle":"2022-08-08T17:20:10.070510Z","shell.execute_reply.started":"2022-08-08T17:20:10.038430Z","shell.execute_reply":"2022-08-08T17:20:10.069775Z"},"trusted":true},"execution_count":null,"outputs":[]}]}