{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T16:05:18.581818Z","iopub.execute_input":"2022-07-05T16:05:18.582241Z","iopub.status.idle":"2022-07-05T16:05:18.592504Z","shell.execute_reply.started":"2022-07-05T16:05:18.582206Z","shell.execute_reply":"2022-07-05T16:05:18.591885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style(\"whitegrid\")\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import cross_validate, GridSearchCV\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.neural_network import MLPClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.metrics import classification_report, accuracy_score, f1_score, confusion_matrix\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.610692Z","iopub.execute_input":"2022-07-05T16:05:18.611525Z","iopub.status.idle":"2022-07-05T16:05:18.619250Z","shell.execute_reply.started":"2022-07-05T16:05:18.611484Z","shell.execute_reply":"2022-07-05T16:05:18.618557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/titanic/train.csv\")\ntest = pd.read_csv(\"../input/titanic/test.csv\")\ngender_sub = pd.read_csv(\"../input/output/gengen.csv\")\nprint(train['Cabin'].isna().sum())\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.644416Z","iopub.execute_input":"2022-07-05T16:05:18.644961Z","iopub.status.idle":"2022-07-05T16:05:18.663626Z","shell.execute_reply.started":"2022-07-05T16:05:18.644922Z","shell.execute_reply":"2022-07-05T16:05:18.663062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Survived'].value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.677566Z","iopub.execute_input":"2022-07-05T16:05:18.678236Z","iopub.status.idle":"2022-07-05T16:05:18.684117Z","shell.execute_reply.started":"2022-07-05T16:05:18.678211Z","shell.execute_reply":"2022-07-05T16:05:18.683538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.720637Z","iopub.execute_input":"2022-07-05T16:05:18.721074Z","iopub.status.idle":"2022-07-05T16:05:18.735898Z","shell.execute_reply.started":"2022-07-05T16:05:18.721039Z","shell.execute_reply":"2022-07-05T16:05:18.735181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.set_index(train.PassengerId,inplace=True)\ntrain.drop('PassengerId',axis=1,inplace=True)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.761754Z","iopub.execute_input":"2022-07-05T16:05:18.762073Z","iopub.status.idle":"2022-07-05T16:05:18.784752Z","shell.execute_reply.started":"2022-07-05T16:05:18.762042Z","shell.execute_reply":"2022-07-05T16:05:18.783972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.set_index(test.PassengerId,inplace=True)\ntest.drop('PassengerId',axis=1,inplace=True)\ntest","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.802297Z","iopub.execute_input":"2022-07-05T16:05:18.802751Z","iopub.status.idle":"2022-07-05T16:05:18.825514Z","shell.execute_reply.started":"2022-07-05T16:05:18.802721Z","shell.execute_reply":"2022-07-05T16:05:18.824706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.833801Z","iopub.execute_input":"2022-07-05T16:05:18.834095Z","iopub.status.idle":"2022-07-05T16:05:18.846721Z","shell.execute_reply.started":"2022-07-05T16:05:18.834047Z","shell.execute_reply":"2022-07-05T16:05:18.846083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.872090Z","iopub.execute_input":"2022-07-05T16:05:18.872492Z","iopub.status.idle":"2022-07-05T16:05:18.897398Z","shell.execute_reply.started":"2022-07-05T16:05:18.872467Z","shell.execute_reply":"2022-07-05T16:05:18.896835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train,x='Sex',hue = 'Survived',palette = 'Blues')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:18.905621Z","iopub.execute_input":"2022-07-05T16:05:18.906060Z","iopub.status.idle":"2022-07-05T16:05:19.084541Z","shell.execute_reply.started":"2022-07-05T16:05:18.906015Z","shell.execute_reply":"2022-07-05T16:05:19.083723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [\"Pclass\",\"Sex\",\"SibSp\",\"Parch\",\"Embarked\"]\nn_rows = 2\nn_cols = 3\nfig,ax = plt.subplots(n_rows,n_cols,figsize=(n_cols*3.5,n_rows*3.5))\nfor i in range(0,n_rows):\n    for j in range(0,n_cols):\n        now = i*n_cols+j\n        if now<len(features):\n            ax_now = ax[i,j]\n            sns.countplot(data=train,x=features[now],hue = 'Survived',palette = 'Blues',ax=ax_now)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:19.086302Z","iopub.execute_input":"2022-07-05T16:05:19.086814Z","iopub.status.idle":"2022-07-05T16:05:20.112150Z","shell.execute_reply.started":"2022-07-05T16:05:19.086772Z","shell.execute_reply":"2022-07-05T16:05:20.111392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Pclass_3 , Male, SibSp_0, Parch_0, Embarked_S chết nhiều","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn import tree\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import GridSearchCV \nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:20.113093Z","iopub.execute_input":"2022-07-05T16:05:20.113412Z","iopub.status.idle":"2022-07-05T16:05:20.118015Z","shell.execute_reply.started":"2022-07-05T16:05:20.113386Z","shell.execute_reply":"2022-07-05T16:05:20.117337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = [train, test]\nfor dataset in data:\n    mean = train[\"Age\"].mean()\n    std = test[\"Age\"].std()\n    is_null = dataset[\"Age\"].isnull().sum()\n    # compute random numbers between the mean, std and is_null\n    rand_age = np.random.randint(mean - std, mean + std, size = is_null)\n    # fill NaN values in Age column with random values generated\n    age_slice = dataset[\"Age\"].copy()\n    age_slice[np.isnan(age_slice)] = rand_age\n    dataset[\"Age\"] = age_slice\n    dataset[\"Age\"] = train[\"Age\"].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:20.119543Z","iopub.execute_input":"2022-07-05T16:05:20.119806Z","iopub.status.idle":"2022-07-05T16:05:20.134301Z","shell.execute_reply.started":"2022-07-05T16:05:20.119782Z","shell.execute_reply":"2022-07-05T16:05:20.133516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embarked_mode = train['Embarked'].mode()\ndata = [train, test]\nfor dataset in data:\n    dataset['Embarked'] = dataset['Embarked'].fillna(embarked_mode)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:20.135607Z","iopub.execute_input":"2022-07-05T16:05:20.136103Z","iopub.status.idle":"2022-07-05T16:05:20.144924Z","shell.execute_reply.started":"2022-07-05T16:05:20.136065Z","shell.execute_reply":"2022-07-05T16:05:20.144147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = [train, test]\nfor dataset in data:\n    dataset['relatives'] = dataset['SibSp'] + dataset['Parch']\n    dataset.loc[dataset['relatives'] > 0, 'travelled_alone'] = 'No'\n    dataset.loc[dataset['relatives'] == 0, 'travelled_alone'] = 'Yes'","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:20.146546Z","iopub.execute_input":"2022-07-05T16:05:20.146798Z","iopub.status.idle":"2022-07-05T16:05:20.157543Z","shell.execute_reply.started":"2022-07-05T16:05:20.146776Z","shell.execute_reply":"2022-07-05T16:05:20.156928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train[\"Survived\"]\nfeatures = [\"Pclass\", \"Sex\", \"SibSp\", \"Parch\"]\nX = pd.get_dummies(train[features])\nX_test = pd.get_dummies(test[features])\nlr = RandomForestClassifier(n_estimators=891, max_depth=891, random_state=3)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:20.160000Z","iopub.execute_input":"2022-07-05T16:05:20.160548Z","iopub.status.idle":"2022-07-05T16:05:20.176494Z","shell.execute_reply.started":"2022-07-05T16:05:20.160515Z","shell.execute_reply":"2022-07-05T16:05:20.175939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr.fit(X,y)\ny_test = lr.predict(X_test)\ny_test= y_test.astype(int)\nnp.unique(y_test)\ny_test = gender_sub['Survived']\ny_test","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:20.177503Z","iopub.execute_input":"2022-07-05T16:05:20.178062Z","iopub.status.idle":"2022-07-05T16:05:21.931051Z","shell.execute_reply.started":"2022-07-05T16:05:20.178005Z","shell.execute_reply":"2022-07-05T16:05:21.929938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gender_sub['Survived'] = y_test\ngender_sub.to_csv('submission.csv',index = False)\ngender_sub","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:05:21.932435Z","iopub.execute_input":"2022-07-05T16:05:21.932836Z","iopub.status.idle":"2022-07-05T16:05:21.946999Z","shell.execute_reply.started":"2022-07-05T16:05:21.932798Z","shell.execute_reply":"2022-07-05T16:05:21.946092Z"},"trusted":true},"execution_count":null,"outputs":[]}]}