{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a550c92831a0c439afb2ea19f9532a6caca61481"},"cell_type":"markdown","source":"> ## Loading Data in python"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df = pd.read_csv(\"../input/train.csv\")\nprint(df.head())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e7ddcdfc38d69a79bab5e5aabf7bb89e9ebd281d"},"cell_type":"markdown","source":"> ## Dropping unecessary columns"},{"metadata":{"trusted":true,"_uuid":"ff2f3bdb031179938d008a94bb4e569c6e25ea47"},"cell_type":"code","source":"y = df['Survived']\nX=df.drop(['Name','Ticket','Survived','PassengerId','Cabin'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"58f943cb90d05193f3e0be9535736c37da3eb09e"},"cell_type":"markdown","source":"> ## Getting dummy columns"},{"metadata":{"trusted":true,"_uuid":"b1d9e13afcb7c3832e60ff12617a6c53530bccea"},"cell_type":"code","source":"X=pd.get_dummies(X,columns=['Sex','Embarked'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"81a593698b774665b363ab155570bc18ae43c325"},"cell_type":"markdown","source":"> ## Creating age categories"},{"metadata":{"trusted":true,"_uuid":"61f5c7cd715dd3943c95ec6f2d00ca9625effb17"},"cell_type":"code","source":"def age_div(x):\n    if x>=0 and x<=8:\n        return 'Infant'\n    elif x>8 and x<=18:\n        return 'Children'\n    elif x>18 and x<=50:\n        return 'Adult'\n    elif x>50:\n        return 'Old'\n    else:\n        return 'NoEntry'\nX['Age_new']=X['Age'].apply(age_div)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"78241a4bbbd40023075cdbb9a07a4a4682aa160b"},"cell_type":"code","source":"X=pd.get_dummies(X,columns=['Age_new'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"732ebdc6c748e1bbe1b7622a368bffd2388fc5db"},"cell_type":"code","source":"X.drop(['Age'],axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4f4d3a666154a7dcfd54c4f97b632d1905cdcbb2"},"cell_type":"code","source":"X.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"060cd940e3aba67a9cc28867c2e0144c4e20cb9e"},"cell_type":"markdown","source":">> ## Train Test Split"},{"metadata":{"trusted":true,"_uuid":"906960ac28089fd61ce9d5a63af3564f90a1c83a"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"154cdf29ed26d891ec7f4ee6909075cff829fb43"},"cell_type":"markdown","source":"> ## Logistic Regression"},{"metadata":{"trusted":true,"_uuid":"c0ea11e591851f239029ea24416acd4c99732a43"},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"31466e403b2b5f3ac4c2eff5cbcedf1aa5d1e3e0"},"cell_type":"code","source":"clf = LogisticRegression(solver='lbfgs',multi_class='multinomial',max_iter=250)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d3338a0ba0e08759bf6ce605b69bfeadc1999e45"},"cell_type":"code","source":"clf.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9005107d399d2c9ea3a957439e4d21e88f0c927e"},"cell_type":"markdown","source":">> ## Predicting Test Set"},{"metadata":{"trusted":true,"_uuid":"5acd7c09f908013bf42de9da0f2012f7426843fc"},"cell_type":"code","source":"y_predict = clf.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d04c8520b3cfd2c1f4c5cf0008ef0b0c270c8ae"},"cell_type":"code","source":"from sklearn.metrics import accuracy_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5abcfed0b685c650e37919002cc15928ce370df2"},"cell_type":"code","source":"acc = accuracy_score(y_test, y_predict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dcba4e1b2436d216749cf6a06f3ebf0fc530e666"},"cell_type":"code","source":"print(\"The accuracy is \",acc*100, \"%.\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6831993c4921234af1b2cd04f15436ce74e6e89f"},"cell_type":"code","source":"df_test = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"700b2d552a2c314c2c83577ac751baa88293cfa4"},"cell_type":"code","source":"X=df_test.drop(['Name','Ticket','PassengerId','Cabin'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce3df23e2c2c3201f93c7aaf02f8bd4bc943fdfa"},"cell_type":"code","source":"def age_div(x):\n    if x>=0 and x<=8:\n        return 'Infant'\n    elif x>8 and x<=18:\n        return 'Children'\n    elif x>18 and x<=50:\n        return 'Adult'\n    elif x>50:\n        return 'Old'\n    else:\n        return 'NoEntry'\nX['Age_new']=X['Age'].apply(age_div)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ceb1fecb9e8b1932f9dba3dffce24fbfa27e1d04"},"cell_type":"code","source":"X=pd.get_dummies(X,columns=['Sex','Embarked','Age_new'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b97ec412d49b983beb90a6b5ecb39f0b6384993"},"cell_type":"code","source":"X.drop(['Age'],axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e570cef8a881e673ce886a6f84c1bccfa0c306a2"},"cell_type":"code","source":"X.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91be265221d160782caee6c94c60179d843d91ee"},"cell_type":"code","source":"X=X.fillna(method='ffill')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b289b28ffdc8eeec57c75003323aced33ccf90fd"},"cell_type":"code","source":"y_pred=clf.predict(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7951806460125170c0021cf44de4aee65d37b76"},"cell_type":"code","source":"serial = df_test['PassengerId']\ndata = {'PassengerId': df_test['PassengerId'], 'Survived': y_pred}\nsubmission = pd.DataFrame(data)\nsubmission.to_csv('Submission.csv', index=False)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb73f808de12dcbb135dd7aefbd4525af826a56d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}