{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T20:47:54.674831Z","iopub.execute_input":"2022-07-13T20:47:54.675484Z","iopub.status.idle":"2022-07-13T20:47:54.716195Z","shell.execute_reply.started":"2022-07-13T20:47:54.675329Z","shell.execute_reply":"2022-07-13T20:47:54.715362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Importing additional necessary modules for this project\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.impute import SimpleImputer\nprint(\"Modules imported\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:54.717730Z","iopub.execute_input":"2022-07-13T20:47:54.718162Z","iopub.status.idle":"2022-07-13T20:47:56.259588Z","shell.execute_reply.started":"2022-07-13T20:47:54.718116Z","shell.execute_reply":"2022-07-13T20:47:56.257265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading train data\ntrain = pd.read_csv('../input/titanic/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:56.261551Z","iopub.execute_input":"2022-07-13T20:47:56.262263Z","iopub.status.idle":"2022-07-13T20:47:56.310167Z","shell.execute_reply.started":"2022-07-13T20:47:56.262210Z","shell.execute_reply":"2022-07-13T20:47:56.309300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading test data\ntest = pd.read_csv('../input/titanic/test.csv')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:56.311725Z","iopub.execute_input":"2022-07-13T20:47:56.312690Z","iopub.status.idle":"2022-07-13T20:47:56.338610Z","shell.execute_reply.started":"2022-07-13T20:47:56.312650Z","shell.execute_reply":"2022-07-13T20:47:56.337492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking for missing values\nprint(train.isnull().sum())\nprint(test.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:56.340413Z","iopub.execute_input":"2022-07-13T20:47:56.340918Z","iopub.status.idle":"2022-07-13T20:47:56.355751Z","shell.execute_reply.started":"2022-07-13T20:47:56.340854Z","shell.execute_reply":"2022-07-13T20:47:56.354431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Handling missing values\ntrain[\"Age\"].fillna(train[\"Age\"].median(), inplace = True)\ntrain[\"Embarked\"].fillna('S',inplace = True)\ntest[\"Age\"].fillna(test[\"Age\"].median(), inplace = True)\ntest[\"Fare\"].fillna(test[\"Fare\"].median(), inplace = True)\nprint(\"Dealt with\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:56.357714Z","iopub.execute_input":"2022-07-13T20:47:56.358234Z","iopub.status.idle":"2022-07-13T20:47:56.379850Z","shell.execute_reply.started":"2022-07-13T20:47:56.358185Z","shell.execute_reply":"2022-07-13T20:47:56.378584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking for missing values again\nprint(train.isnull().sum())\nprint(test.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:56.381843Z","iopub.execute_input":"2022-07-13T20:47:56.382691Z","iopub.status.idle":"2022-07-13T20:47:56.396515Z","shell.execute_reply.started":"2022-07-13T20:47:56.382643Z","shell.execute_reply":"2022-07-13T20:47:56.395488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Manipulating data\ntrain['Male'] = train['Sex'] == 'male'\ntest['Male'] = test['Sex'] == 'male'\ntrain['Embarked'] = train['Embarked'].replace({'S':0,'C':1,'Q':2})\ntest['Embarked'] = test['Embarked'].replace({'S':0,'C':1,'Q':2})\n\nX = train[['Pclass','Male','Fare','Age','SibSp','Embarked','Parch']].values\ntest_X = test[['Pclass','Male','Fare','Age','SibSp','Embarked','Parch']].values\ny = train['Survived'].values\nprint(X)\nprint(test_X)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:56.398601Z","iopub.execute_input":"2022-07-13T20:47:56.399396Z","iopub.status.idle":"2022-07-13T20:47:56.428644Z","shell.execute_reply.started":"2022-07-13T20:47:56.399346Z","shell.execute_reply":"2022-07-13T20:47:56.426926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Training model\nmodel = LogisticRegression()\nmodel.fit(X,y)\nmodel.score(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:56.434914Z","iopub.execute_input":"2022-07-13T20:47:56.435661Z","iopub.status.idle":"2022-07-13T20:47:56.508746Z","shell.execute_reply.started":"2022-07-13T20:47:56.435577Z","shell.execute_reply":"2022-07-13T20:47:56.507312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making prediction\nresult = model.predict(test_X)\nprediction = test[['PassengerId']]\nprediction.insert(1,'Survived',result,True)\nprediction.to_csv('./submission.csv',index=False)\nprint (\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:47:56.566099Z","iopub.execute_input":"2022-07-13T20:47:56.567104Z","iopub.status.idle":"2022-07-13T20:47:56.582521Z","shell.execute_reply.started":"2022-07-13T20:47:56.567059Z","shell.execute_reply":"2022-07-13T20:47:56.581732Z"},"trusted":true},"execution_count":null,"outputs":[]}]}