{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport warnings #Used to ignore the warning given as output of the code.\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import the datasets\n\ntrain_data = pd.read_csv('../input/titanic/train.csv')\ntest_data = pd.read_csv('../input/titanic/test.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inspecting the data","metadata":{}},{"cell_type":"code","source":"def analyze(df):\n    male, female = df.Sex.value_counts()\n    n = len(df.axes[0])\n\n    print(f\"The row count is \" + str(n))\n    print(f\"The percent of male to female is \" + str(male / female))\n    print(f\"The percent of nan in Cabin is \" + str((df.Cabin.isna().sum() - 1) / n))\n    print(f\"The percent of nan in Age is \" + str((df.Age.isna().sum() - 1) / n))\n    print(f\"The mean age is \" + str(df.Age.mean()))\n    print(f\"The mean fare cost is \" + str(df.Fare.mean()))\n    print(f\"The count of classes are\\n\" + str(df.Pclass.value_counts()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_copy = train_data.copy()\ntest_data_copy = test_data.copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"# instead of bootstrapping the data which would introduce centrality and bias\n# into the model (I was going to set all NaN to the mean Age)\n# I decided to drop the rows that have NaN entries for the families age\n# this dropped ~20% of the total records\n\nprint(f'Train Data Record Count ' + str(len(train_data_copy.axes[0])))\nprint(f'Test Data Record Count ' + str(len(test_data_copy.axes[0])))\n\ntrain_data = train_data[train_data[\"Age\"].notna()]\ntest_data = test_data[test_data[\"Age\"].notna()]\n\n# EXPLAIN WHY WE DROPPED EMPTY EMBARKED BECAUSE THE MODEL COULDNT RUN WITH 2 NaN RECORDS\n\ntrain_data = train_data[train_data[\"Embarked\"].notna()]\ntest_data = test_data[test_data[\"Embarked\"].notna()]\n\nprint(f'Train Data Record Count ' + str(len(train_data.axes[0])))\nprint(f'Test Data Record Count ' + str(len(test_data.axes[0])))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split the training data up so that we can measure the fitness of the models\n\nfeatures = {\"Survived\",\"Age\",\"Sex\",\"Pclass\",\"Embarked\"}\n\ntrain_data_edited = train_data[features].reset_index()\n\nX = train_data_edited.drop(\"Survived\", axis=1)\ny = train_data_edited[\"Survived\"]\n\n# fixes datatype errors\nX.Age = X.Age.round(0).astype(int)\n# X.Fare = X.Fare.round(0).astype(int)\n\n# Split features and target into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=.1, random_state=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Building","metadata":{}},{"cell_type":"code","source":"forest = RandomForestClassifier(n_estimators=100)\nforest.fit(pd.get_dummies(X_train), y_train)\ny_pred = forest.predict(pd.get_dummies(X_test))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate Model Accuracy\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Test Data Record Count ' + str(len(test_data_copy.axes[0])))\n\ntest_data_edited = test_data_copy[{\"Age\",\"Sex\",\"Pclass\",\"Embarked\"}].reset_index()\n\ntest_data_edited.Age = test_data_edited.Age.fillna(30)\n\ntest_data_edited = pd.get_dummies(test_data_edited)\n\ntest_data_edited","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Survived = forest.predict(test_data_edited)\noutput = pd.DataFrame({'PassengerId': test_data_copy.PassengerId, 'Survived': Survived})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{},"execution_count":null,"outputs":[]}]}