{"cells":[{"metadata":{"_uuid":"ceede006376cb7a46b2299298b8c77d080a9a594"},"cell_type":"markdown","source":"# Getting started with my first Kaggle entry\n\nThis Kaggle entry was a quick and dirty way to get started. I treid to do the simplest submission I could think of to get the hang of how Kaggle works.  I adapted some of this code from the wonderful book \"Introduction to Machine Learning with Python\" from Oreilly. \n\n\n# Import Python modules to notebook"},{"metadata":{"trusted":true,"_uuid":"4584b01833481dd45614291465f45c7705edf956"},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.ensemble import RandomForestClassifier\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7c9a6d5177d2feed9b8d2c19a4dc88046aea6555"},"cell_type":"markdown","source":"# Get the input data"},{"metadata":{"trusted":true,"_uuid":"3dfedce9a7da9bdd5720a60317d9bbe3306890a4"},"cell_type":"code","source":"df_raw = pd.read_csv('../input/train.csv', low_memory=False)\ndf_raw.head(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b769ac9afe1e7dab46fa201f00e0fe8b0bfafbf9"},"cell_type":"markdown","source":"# Drop the columns that dont help predict the outcome"},{"metadata":{"trusted":true,"_uuid":"2d8b9d23c331c5c803d5e51d402d201454f385cb"},"cell_type":"code","source":"df_drop = df_raw.drop(columns=['Name','Ticket','Cabin','PassengerId'])\ndf_drop.head(5)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8117048c5160a0c0b8b3bccc66c858155d8e7baf"},"cell_type":"markdown","source":"# Encode text columns because models need numbers not text"},{"metadata":{"trusted":true,"_uuid":"b47841b8d9a256a10c296d684334515ec5b4ad75"},"cell_type":"code","source":"df_dummies = pd.get_dummies(df_drop,columns=['Pclass','Sex','Embarked'])\ndf_dummies.head(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5c00558872b0d7e109995ac306e4392cf5aed689"},"cell_type":"markdown","source":"# Split out target column from data\n\nThis code separates the survived column from the data to feed into the train_test_split function.  I also needed to fix the NaN values in the dataset by replacing them with the mean of the data in the column."},{"metadata":{"trusted":true,"_uuid":"f4d13da7e4856e398c265fcfd47417e223e96588"},"cell_type":"code","source":"df_target = df_dummies['Survived']\ndf_data = df_dummies.drop(columns=['Survived'])\n# fill in NaN values\ndf_data.fillna(df_data.mean(), inplace=True)\n\nX_train, X_test, y_train, y_test = train_test_split(df_data, df_target, random_state=0)\ndf_data.head(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b1578aacec97af9aade6bd56f47a580fe40d81a4"},"cell_type":"markdown","source":"# Create KNeighbors classifier to see how it does"},{"metadata":{"trusted":true,"_uuid":"1c7ed2e2384ea738e20b3666cacc583fb7e608db"},"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors=6)\nknn.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0ee0a28ee149c7fbb408017b2934aded8da0065"},"cell_type":"code","source":"print(\"Test set score: {:.2f}\".format(knn.score(X_test, y_test)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e8a6e161b8937518ba5405d60d2695e894d1ac88"},"cell_type":"markdown","source":"# Create Random Forest Classisifer to see how it does"},{"metadata":{"trusted":true,"_uuid":"03227e28a3486fd21703c86f451dccfadadabda1"},"cell_type":"code","source":"forest = RandomForestClassifier(n_estimators=5, random_state=2)\nforest.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"150ca6e51c2fdaad1dfd9e01a6c27e6b8bd8c481"},"cell_type":"code","source":"print(\"Test set score: {:.2f}\".format(forest.score(X_test, y_test)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"564c0a016d4349bcba8f0186abdea09caab00604"},"cell_type":"markdown","source":"# Load the actual test data and get some predicitons using the Random forest because it was better"},{"metadata":{"trusted":true,"_uuid":"e9743010fe7f0ef7723f58e9bd0519b81357d40f"},"cell_type":"code","source":"df_test = pd.read_csv('../input/test.csv', low_memory=False)\ndf_test_drop = df_test.drop(columns=['Name','Ticket','Cabin','PassengerId'])\ndf_test_dummies = pd.get_dummies(df_test_drop,columns=['Pclass','Sex','Embarked'])\n# fill in NaN values\ndf_test_dummies.fillna(df_test_dummies.mean(), inplace=True)\npredictions = forest.predict(df_test_dummies)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"249efeb5c4840c93c505fb8bae93b737e0fb3434"},"cell_type":"code","source":"predictions","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"74d055602ff668639fa30458a2af45fab273119d"},"cell_type":"markdown","source":"# Create a data frame and join the passenger id to the predictions.  Create CSV for submittal"},{"metadata":{"trusted":true,"_uuid":"cec48204c3047371875d651b9c0fd2eb037f0b93"},"cell_type":"code","source":"submission = pd.DataFrame({'Passengerid': df_test.PassengerId, 'Survived': predictions})\n#submission.to_csv(\"../output/titanicsubmit.csv\", index=False)\nsubmission.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d434da3aa84dd5bce2a5fc3b4dae83ed124cb8f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}