{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"****Titanic:74%**** Not too bad for the first competetions, is it ? :) \nI will be so thankful for any advices, be sure  your comment will make an improvement So don't hesitate to share't .","metadata":{}},{"cell_type":"markdown","source":"**Importing libraries**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T16:16:44.705969Z","iopub.execute_input":"2022-07-05T16:16:44.706326Z","iopub.status.idle":"2022-07-05T16:16:44.714401Z","shell.execute_reply.started":"2022-07-05T16:16:44.706296Z","shell.execute_reply":"2022-07-05T16:16:44.713270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Importing the dataset**","metadata":{}},{"cell_type":"code","source":"train =pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest=pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntest_ids = test[\"PassengerId\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:16:48.664103Z","iopub.execute_input":"2022-07-05T16:16:48.664628Z","iopub.status.idle":"2022-07-05T16:16:48.684686Z","shell.execute_reply.started":"2022-07-05T16:16:48.664584Z","shell.execute_reply":"2022-07-05T16:16:48.683645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:16:53.163844Z","iopub.execute_input":"2022-07-05T16:16:53.164724Z","iopub.status.idle":"2022-07-05T16:16:53.184106Z","shell.execute_reply.started":"2022-07-05T16:16:53.164681Z","shell.execute_reply":"2022-07-05T16:16:53.182917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dataset information (sweetviz)**","metadata":{}},{"cell_type":"code","source":"!pip install sweetviz\n#creating a EDA report\nimport sweetviz as sv\nanalyze_report = sv.analyze(test)\nanalyze_report.show_html('analyze.html', open_browser=False)\n \n\nfrom IPython.display import IFrame\nIFrame(src = 'analyze.html',width=1000,height=600)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:17:10.394676Z","iopub.execute_input":"2022-07-05T16:17:10.395855Z","iopub.status.idle":"2022-07-05T16:17:25.801146Z","shell.execute_reply.started":"2022-07-05T16:17:10.395776Z","shell.execute_reply":"2022-07-05T16:17:25.799931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dropping unnecessary columns and Taking care of misssing data** ","metadata":{}},{"cell_type":"code","source":"def clean(data):\n    data = data.drop([\"Ticket\", \"PassengerId\", \"Name\", \"Cabin\"], axis=1)\n    \n    cols = [\"SibSp\", \"Parch\", \"Fare\", \"Age\"]\n    for col in cols:\n        data[col].fillna(data[col].median(), inplace=True)\n        \n    data.Embarked.fillna(\"S\", inplace=True)\n    return data\n\ntrain = clean(train)\ntest = clean(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.347896Z","iopub.execute_input":"2022-07-05T16:12:01.348661Z","iopub.status.idle":"2022-07-05T16:12:01.367752Z","shell.execute_reply.started":"2022-07-05T16:12:01.348616Z","shell.execute_reply":"2022-07-05T16:12:01.366076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.370077Z","iopub.execute_input":"2022-07-05T16:12:01.371003Z","iopub.status.idle":"2022-07-05T16:12:01.388014Z","shell.execute_reply.started":"2022-07-05T16:12:01.370953Z","shell.execute_reply":"2022-07-05T16:12:01.386346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Encoding categorical data**","metadata":{}},{"cell_type":"code","source":"from sklearn import preprocessing\nle = preprocessing.LabelEncoder()\ncolumns = [\"Sex\", \"Embarked\"]\n\nfor col in columns:\n    train[col] = le.fit_transform(train[col])\n    test[col] = le.transform(test[col])\n    print(le.classes_)\n      \ntrain.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.392409Z","iopub.execute_input":"2022-07-05T16:12:01.393342Z","iopub.status.idle":"2022-07-05T16:12:01.418752Z","shell.execute_reply.started":"2022-07-05T16:12:01.393254Z","shell.execute_reply":"2022-07-05T16:12:01.417964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Spliting the Train & Test datasets**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\n\ny = train[\"Survived\"]\nX = train.drop(\"Survived\", axis=1)\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.420012Z","iopub.execute_input":"2022-07-05T16:12:01.420777Z","iopub.status.idle":"2022-07-05T16:12:01.429671Z","shell.execute_reply.started":"2022-07-05T16:12:01.420743Z","shell.execute_reply":"2022-07-05T16:12:01.428703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Initiate  the model**","metadata":{}},{"cell_type":"code","source":"clf = LogisticRegression(random_state=0, max_iter=1000).fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.431433Z","iopub.execute_input":"2022-07-05T16:12:01.432134Z","iopub.status.idle":"2022-07-05T16:12:01.47462Z","shell.execute_reply.started":"2022-07-05T16:12:01.432092Z","shell.execute_reply":"2022-07-05T16:12:01.473589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Make predictions and evaluate the model**","metadata":{}},{"cell_type":"code","source":"predictions = clf.predict(X_val)\nfrom sklearn.metrics import accuracy_score\naccuracy_score(y_val, predictions)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.476274Z","iopub.execute_input":"2022-07-05T16:12:01.476896Z","iopub.status.idle":"2022-07-05T16:12:01.487252Z","shell.execute_reply.started":"2022-07-05T16:12:01.476861Z","shell.execute_reply":"2022-07-05T16:12:01.486161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Predict the test csv**","metadata":{}},{"cell_type":"code","source":"submission_preds = clf.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.48878Z","iopub.execute_input":"2022-07-05T16:12:01.489417Z","iopub.status.idle":"2022-07-05T16:12:01.49637Z","shell.execute_reply.started":"2022-07-05T16:12:01.489387Z","shell.execute_reply":"2022-07-05T16:12:01.495308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submission file output**","metadata":{}},{"cell_type":"code","source":"df = pd.DataFrame({\"PassengerId\": test_ids.values,\n                   \"Survived\": submission_preds,\n                  })","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.498003Z","iopub.execute_input":"2022-07-05T16:12:01.498688Z","iopub.status.idle":"2022-07-05T16:12:01.50733Z","shell.execute_reply.started":"2022-07-05T16:12:01.498642Z","shell.execute_reply":"2022-07-05T16:12:01.506229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:12:01.510057Z","iopub.execute_input":"2022-07-05T16:12:01.510616Z","iopub.status.idle":"2022-07-05T16:12:01.521363Z","shell.execute_reply.started":"2022-07-05T16:12:01.510572Z","shell.execute_reply":"2022-07-05T16:12:01.520186Z"},"trusted":true},"execution_count":null,"outputs":[]}]}