{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:38.839552Z","iopub.execute_input":"2022-07-25T08:51:38.840944Z","iopub.status.idle":"2022-07-25T08:51:38.845335Z","shell.execute_reply.started":"2022-07-25T08:51:38.840889Z","shell.execute_reply":"2022-07-25T08:51:38.844522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"../input/titanic/train.csv\")\ntest = pd.read_csv(\"../input/titanic/test.csv\")\ntest_ids  = test[\"PassengerId\"]\n\ndef clean(data):\n    data = data.drop([\"Ticket\",\"Cabin\", \"Name\", \"PassengerId\"], axis=1)\n    \n    columns = [\"SibSp\", \"Parch\", \"Fare\", \"Age\"]\n    for col in columns:\n        data[col].fillna(data[col].median(), inplace=True)\n        \n    data.Embarked.fillna(\"U\", inplace=True)\n    \n    return data\n\ndata = clean(data)\ntest = clean(test)\n\ndata.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T08:51:38.901997Z","iopub.execute_input":"2022-07-25T08:51:38.902350Z","iopub.status.idle":"2022-07-25T08:51:38.941436Z","shell.execute_reply.started":"2022-07-25T08:51:38.902321Z","shell.execute_reply":"2022-07-25T08:51:38.940653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\nle = preprocessing.LabelEncoder()\n\ncolumns = [\"Sex\", \"Embarked\"]\nfor col in columns:\n    data[col] = le.fit_transform(data[col])\n    test[col] = le.transform(test[col])\n    print(le.classes_)\n    \ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:38.949325Z","iopub.execute_input":"2022-07-25T08:51:38.949679Z","iopub.status.idle":"2022-07-25T08:51:38.968989Z","shell.execute_reply.started":"2022-07-25T08:51:38.949648Z","shell.execute_reply":"2022-07-25T08:51:38.968083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\n\ny = data[\"Survived\"]\nx = data.drop(\"Survived\", axis=1)\n\nx_train, x_val, y_train, y_val = train_test_split(x, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:38.981439Z","iopub.execute_input":"2022-07-25T08:51:38.981822Z","iopub.status.idle":"2022-07-25T08:51:38.991085Z","shell.execute_reply.started":"2022-07-25T08:51:38.981790Z","shell.execute_reply":"2022-07-25T08:51:38.989936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = LogisticRegression(random_state=0, max_iter=1000).fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:39.011596Z","iopub.execute_input":"2022-07-25T08:51:39.012456Z","iopub.status.idle":"2022-07-25T08:51:39.053422Z","shell.execute_reply.started":"2022-07-25T08:51:39.012410Z","shell.execute_reply":"2022-07-25T08:51:39.052281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = clf.predict(x_val)\nfrom sklearn.metrics import accuracy_score\naccuracy_score(y_val, predictions)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:39.055030Z","iopub.execute_input":"2022-07-25T08:51:39.055342Z","iopub.status.idle":"2022-07-25T08:51:39.067581Z","shell.execute_reply.started":"2022-07-25T08:51:39.055312Z","shell.execute_reply":"2022-07-25T08:51:39.066205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_preds = clf.predict(test)\ndf = pd.DataFrame({\"PassengerId\": test_ids.values, \n                    \"Survived\": submission_preds\n                  })","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:39.070161Z","iopub.execute_input":"2022-07-25T08:51:39.070488Z","iopub.status.idle":"2022-07-25T08:51:39.079342Z","shell.execute_reply.started":"2022-07-25T08:51:39.070458Z","shell.execute_reply":"2022-07-25T08:51:39.078224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save file\ndf.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:51:39.091230Z","iopub.execute_input":"2022-07-25T08:51:39.091952Z","iopub.status.idle":"2022-07-25T08:51:39.099876Z","shell.execute_reply.started":"2022-07-25T08:51:39.091920Z","shell.execute_reply":"2022-07-25T08:51:39.098814Z"},"trusted":true},"execution_count":null,"outputs":[]}]}