{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder, StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.linear_model import LogisticRegression\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T15:13:16.755036Z","iopub.execute_input":"2022-07-27T15:13:16.755493Z","iopub.status.idle":"2022-07-27T15:13:16.766089Z","shell.execute_reply.started":"2022-07-27T15:13:16.755459Z","shell.execute_reply":"2022-07-27T15:13:16.764853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/titanic/train.csv')\ndf_y = df[['Survived']]\ndf_x = df.drop(columns=['Survived'])\ndf_test = pd.read_csv('/kaggle/input/titanic/test.csv')\nprint(\"Dataset Shape:\", df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:16.771125Z","iopub.execute_input":"2022-07-27T15:13:16.771528Z","iopub.status.idle":"2022-07-27T15:13:16.824062Z","shell.execute_reply.started":"2022-07-27T15:13:16.771492Z","shell.execute_reply":"2022-07-27T15:13:16.822870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_x.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:16.826456Z","iopub.execute_input":"2022-07-27T15:13:16.827713Z","iopub.status.idle":"2022-07-27T15:13:16.843892Z","shell.execute_reply.started":"2022-07-27T15:13:16.827661Z","shell.execute_reply":"2022-07-27T15:13:16.842294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_x.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:16.845822Z","iopub.execute_input":"2022-07-27T15:13:16.846823Z","iopub.status.idle":"2022-07-27T15:13:16.876218Z","shell.execute_reply.started":"2022-07-27T15:13:16.846772Z","shell.execute_reply":"2022-07-27T15:13:16.875070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in df:\n  print(col, len(df[col].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:16.879079Z","iopub.execute_input":"2022-07-27T15:13:16.879701Z","iopub.status.idle":"2022-07-27T15:13:16.892220Z","shell.execute_reply.started":"2022-07-27T15:13:16.879668Z","shell.execute_reply":"2022-07-27T15:13:16.890814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_x.drop(columns=['Name', 'Ticket', 'Cabin', 'PassengerId'], inplace=True)\ndf_x.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:16.893899Z","iopub.execute_input":"2022-07-27T15:13:16.895070Z","iopub.status.idle":"2022-07-27T15:13:16.926647Z","shell.execute_reply.started":"2022-07-27T15:13:16.895021Z","shell.execute_reply":"2022-07-27T15:13:16.924922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\nsex_enc = le.fit_transform(df_x['Sex'])\ndf_x['Sex'] = sex_enc # male:1 ,female:0\ndf_x.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:16.928585Z","iopub.execute_input":"2022-07-27T15:13:16.929065Z","iopub.status.idle":"2022-07-27T15:13:16.946830Z","shell.execute_reply.started":"2022-07-27T15:13:16.929020Z","shell.execute_reply":"2022-07-27T15:13:16.945739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mode = df_x['Embarked'].mode()[0]\ndf_x['Embarked'] = df_x['Embarked'].fillna(mode)\nohe = OneHotEncoder()\nft = ohe.fit_transform(df_x[['Embarked']]).toarray()\nemb_enc_df = pd.DataFrame(ft, columns=['C', 'Q', 'S'])\ndf_x = df_x.join(emb_enc_df)\ndf_x.drop(columns=['Embarked'], inplace=True)\ndf_x.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:16.948500Z","iopub.execute_input":"2022-07-27T15:13:16.949180Z","iopub.status.idle":"2022-07-27T15:13:16.981166Z","shell.execute_reply.started":"2022-07-27T15:13:16.949138Z","shell.execute_reply":"2022-07-27T15:13:16.980149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"si = SimpleImputer()\nprint(\"Before Imputing:\", df_x.dropna(subset=['Age'])['Age'].mean())\nage_imp = si.fit_transform(df_x[['Age']])\ndf_x['Age'] = age_imp\nprint(\"After Imputing:\", df_x['Age'].mean())\ndisplay(df_x.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:16.982408Z","iopub.execute_input":"2022-07-27T15:13:16.982915Z","iopub.status.idle":"2022-07-27T15:13:17.003734Z","shell.execute_reply.started":"2022-07-27T15:13:16.982885Z","shell.execute_reply":"2022-07-27T15:13:17.002136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_x.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:17.005195Z","iopub.execute_input":"2022-07-27T15:13:17.005681Z","iopub.status.idle":"2022-07-27T15:13:17.025125Z","shell.execute_reply.started":"2022-07-27T15:13:17.005636Z","shell.execute_reply":"2022-07-27T15:13:17.023173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_x.shape, df_y.shape, df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:17.029586Z","iopub.execute_input":"2022-07-27T15:13:17.030722Z","iopub.status.idle":"2022-07-27T15:13:17.044562Z","shell.execute_reply.started":"2022-07-27T15:13:17.030670Z","shell.execute_reply":"2022-07-27T15:13:17.043584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nscaled = scaler.fit_transform(df_x)\ndf_x = pd.DataFrame(scaled, columns=df_x.columns)\ndf_x.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:17.045874Z","iopub.execute_input":"2022-07-27T15:13:17.046654Z","iopub.status.idle":"2022-07-27T15:13:17.071379Z","shell.execute_reply.started":"2022-07-27T15:13:17.046607Z","shell.execute_reply":"2022-07-27T15:13:17.070069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()\ndf_test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:17.072755Z","iopub.execute_input":"2022-07-27T15:13:17.073044Z","iopub.status.idle":"2022-07-27T15:13:17.095193Z","shell.execute_reply.started":"2022-07-27T15:13:17.073018Z","shell.execute_reply":"2022-07-27T15:13:17.094010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_passengers = df_test.PassengerId\ndf_test.drop(columns=['Name', 'Ticket', 'Cabin', 'PassengerId'], inplace=True)\ndf_test['Fare'] = df_test['Fare'].fillna(df_test['Fare'].mode()[0])\ndf_test['Age'] = si.fit_transform(df_test[['Age']])\ndf_test['Sex'] = le.fit_transform(df_test['Sex'])\ndf_test = df_test.join(pd.DataFrame(ohe.fit_transform(df_test[['Embarked']]).toarray(), columns=['C', 'Q', 'S']))\ndf_test.drop(columns=['Embarked'], inplace=True)\nscaled_test = scaler.fit_transform(df_test)\ndf_test = pd.DataFrame(scaled_test, columns=df_test.columns)\ndf_test.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:17.096687Z","iopub.execute_input":"2022-07-27T15:13:17.097689Z","iopub.status.idle":"2022-07-27T15:13:17.145701Z","shell.execute_reply.started":"2022-07-27T15:13:17.097636Z","shell.execute_reply":"2022-07-27T15:13:17.144005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = LogisticRegression()\nlr.fit(df_x, df_y.values.ravel())\nprediction = lr.predict(df_test)\ndct = {\n    \"PassengerId\": df_test_passengers,\n    \"Survived\": prediction\n}\npred_df = pd.DataFrame(dct)\npred_df.to_csv('prediction.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:13:17.147384Z","iopub.execute_input":"2022-07-27T15:13:17.147983Z","iopub.status.idle":"2022-07-27T15:13:17.179057Z","shell.execute_reply.started":"2022-07-27T15:13:17.147947Z","shell.execute_reply":"2022-07-27T15:13:17.177085Z"},"trusted":true},"execution_count":null,"outputs":[]}]}