{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-15T01:48:31.425377Z","iopub.execute_input":"2022-08-15T01:48:31.426467Z","iopub.status.idle":"2022-08-15T01:48:31.435529Z","shell.execute_reply.started":"2022-08-15T01:48:31.426415Z","shell.execute_reply":"2022-08-15T01:48:31.434457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**In this project, I hope to introduce a procedure of Machine Learning project including datapreprocessing, selecting models and fine tuning. Because the dataset is quite small, I ignore validation step for simplicity.**","metadata":{}},{"cell_type":"code","source":"# Load the datasets\ntrain_data = pd.read_csv('/kaggle/input/titanic/train.csv')\ntest_data = pd.read_csv('/kaggle/input/titanic/test.csv')\nval_data = pd.read_csv('/kaggle/input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.437375Z","iopub.execute_input":"2022-08-15T01:48:31.438065Z","iopub.status.idle":"2022-08-15T01:48:31.466895Z","shell.execute_reply.started":"2022-08-15T01:48:31.438030Z","shell.execute_reply":"2022-08-15T01:48:31.465819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Survived'] = val_data['Survived'].copy()\npassengerId = test_data['PassengerId'].copy() # Store passengerId for submission in the last step","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.468151Z","iopub.execute_input":"2022-08-15T01:48:31.468680Z","iopub.status.idle":"2022-08-15T01:48:31.474175Z","shell.execute_reply.started":"2022-08-15T01:48:31.468649Z","shell.execute_reply":"2022-08-15T01:48:31.473220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dataset exploration**","metadata":{}},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.476037Z","iopub.execute_input":"2022-08-15T01:48:31.476576Z","iopub.status.idle":"2022-08-15T01:48:31.506626Z","shell.execute_reply.started":"2022-08-15T01:48:31.476544Z","shell.execute_reply":"2022-08-15T01:48:31.505657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.508664Z","iopub.execute_input":"2022-08-15T01:48:31.509868Z","iopub.status.idle":"2022-08-15T01:48:31.531857Z","shell.execute_reply.started":"2022-08-15T01:48:31.509822Z","shell.execute_reply":"2022-08-15T01:48:31.530754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check null data for train and test data","metadata":{}},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.533658Z","iopub.execute_input":"2022-08-15T01:48:31.534008Z","iopub.status.idle":"2022-08-15T01:48:31.547115Z","shell.execute_reply.started":"2022-08-15T01:48:31.533977Z","shell.execute_reply":"2022-08-15T01:48:31.546346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.549418Z","iopub.execute_input":"2022-08-15T01:48:31.550558Z","iopub.status.idle":"2022-08-15T01:48:31.563884Z","shell.execute_reply.started":"2022-08-15T01:48:31.550515Z","shell.execute_reply":"2022-08-15T01:48:31.562846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Filling the missing value for train data**","metadata":{}},{"cell_type":"markdown","source":"*We can see that  in the train data has 3 feature containing null values which are Age, Cabin and Embarked*","metadata":{}},{"cell_type":"markdown","source":"There are only 2 samples for Embarked which we can delete 2 samples and do not affect too much to our train dataset","metadata":{}},{"cell_type":"code","source":"train_data = train_data.dropna(axis=0, subset=['Embarked'])\ntrain_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.565588Z","iopub.execute_input":"2022-08-15T01:48:31.565922Z","iopub.status.idle":"2022-08-15T01:48:31.588724Z","shell.execute_reply.started":"2022-08-15T01:48:31.565893Z","shell.execute_reply":"2022-08-15T01:48:31.587374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the size of train data after removing NaN row of Embarked\ntrain_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.591517Z","iopub.execute_input":"2022-08-15T01:48:31.592832Z","iopub.status.idle":"2022-08-15T01:48:31.601630Z","shell.execute_reply.started":"2022-08-15T01:48:31.592787Z","shell.execute_reply":"2022-08-15T01:48:31.600553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.603168Z","iopub.execute_input":"2022-08-15T01:48:31.604172Z","iopub.status.idle":"2022-08-15T01:48:31.623696Z","shell.execute_reply.started":"2022-08-15T01:48:31.604138Z","shell.execute_reply":"2022-08-15T01:48:31.622375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We should remove the name and ticket because it does not have meaning for prediction\ntrain_data = train_data.drop(['Name', 'Ticket', 'PassengerId'], axis=1)\ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.628488Z","iopub.execute_input":"2022-08-15T01:48:31.629046Z","iopub.status.idle":"2022-08-15T01:48:31.646050Z","shell.execute_reply.started":"2022-08-15T01:48:31.629012Z","shell.execute_reply":"2022-08-15T01:48:31.644682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*We can see that Cabin have lots of missing value which is not good for prediction. Therefore, we also should remove Cabin feature*","metadata":{}},{"cell_type":"code","source":"train_data = train_data.drop(['Cabin'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.647754Z","iopub.execute_input":"2022-08-15T01:48:31.648435Z","iopub.status.idle":"2022-08-15T01:48:31.659791Z","shell.execute_reply.started":"2022-08-15T01:48:31.648399Z","shell.execute_reply":"2022-08-15T01:48:31.658371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.666487Z","iopub.execute_input":"2022-08-15T01:48:31.666896Z","iopub.status.idle":"2022-08-15T01:48:31.683849Z","shell.execute_reply.started":"2022-08-15T01:48:31.666859Z","shell.execute_reply":"2022-08-15T01:48:31.682767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*For now, we have Age feature which contains missing values. However, it just has a small number of missing value which we can fill by using some methods*\n\nHere, I want to use the the closet non-Nan value to fill the Nan value in Age. However, we have some categorical data which are Sex and Embarked which need to be converted to numerical data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Creating instance for label encoder\nle = LabelEncoder()\n# Transform Sex in train_data to numerical data\nnumerical_Sex = le.fit_transform(train_data['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.685501Z","iopub.execute_input":"2022-08-15T01:48:31.686029Z","iopub.status.idle":"2022-08-15T01:48:31.695950Z","shell.execute_reply.started":"2022-08-15T01:48:31.685996Z","shell.execute_reply":"2022-08-15T01:48:31.694598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To alternate Sex in train_data, we need to remove old data and create a new one\ntrain_data = train_data.drop(['Sex'], axis=1)\ntrain_data['Sex'] = numerical_Sex\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.700719Z","iopub.execute_input":"2022-08-15T01:48:31.701068Z","iopub.status.idle":"2022-08-15T01:48:31.734649Z","shell.execute_reply.started":"2022-08-15T01:48:31.701039Z","shell.execute_reply":"2022-08-15T01:48:31.732997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can do the same actions for Embarked\nnumerical_Embarked = le.fit_transform(train_data['Embarked'])\ntrain_data = train_data.drop(['Embarked'], axis=1)\ntrain_data['Embarked'] = numerical_Embarked\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.736694Z","iopub.execute_input":"2022-08-15T01:48:31.737032Z","iopub.status.idle":"2022-08-15T01:48:31.765166Z","shell.execute_reply.started":"2022-08-15T01:48:31.737002Z","shell.execute_reply":"2022-08-15T01:48:31.763973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.766918Z","iopub.execute_input":"2022-08-15T01:48:31.767311Z","iopub.status.idle":"2022-08-15T01:48:31.780741Z","shell.execute_reply.started":"2022-08-15T01:48:31.767230Z","shell.execute_reply":"2022-08-15T01:48:31.779679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling the missing value of Age with nearest row\nnew_train_data = train_data.interpolate(method='nearest', axis=0)\nnew_train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.782001Z","iopub.execute_input":"2022-08-15T01:48:31.782353Z","iopub.status.idle":"2022-08-15T01:48:31.802673Z","shell.execute_reply.started":"2022-08-15T01:48:31.782321Z","shell.execute_reply":"2022-08-15T01:48:31.801640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We have new_train_data is processed data from train_data with all non-Nan values. Now we need to do the same thing for test dataset**","metadata":{}},{"cell_type":"code","source":"test_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.804530Z","iopub.execute_input":"2022-08-15T01:48:31.804876Z","iopub.status.idle":"2022-08-15T01:48:31.819431Z","shell.execute_reply.started":"2022-08-15T01:48:31.804843Z","shell.execute_reply":"2022-08-15T01:48:31.818257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_data.drop(['Cabin', 'Name', 'Ticket', 'PassengerId'], axis=1)\ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.820860Z","iopub.execute_input":"2022-08-15T01:48:31.821449Z","iopub.status.idle":"2022-08-15T01:48:31.847599Z","shell.execute_reply.started":"2022-08-15T01:48:31.821416Z","shell.execute_reply":"2022-08-15T01:48:31.846364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To alternate Sex in test_data, we need to remove old data and create a new one\nnumerical_Sex = le.fit_transform(test_data['Sex'])\ntest_data = test_data.drop(['Sex'], axis=1)\ntest_data['Sex'] = numerical_Sex\ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.850615Z","iopub.execute_input":"2022-08-15T01:48:31.851063Z","iopub.status.idle":"2022-08-15T01:48:31.867842Z","shell.execute_reply.started":"2022-08-15T01:48:31.851018Z","shell.execute_reply":"2022-08-15T01:48:31.866732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can do the same actions for Embarked\nnumerical_Embarked = le.fit_transform(test_data['Embarked'])\ntest_data = test_data.drop(['Embarked'], axis=1)\ntest_data['Embarked'] = numerical_Embarked","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.868991Z","iopub.execute_input":"2022-08-15T01:48:31.869547Z","iopub.status.idle":"2022-08-15T01:48:31.877951Z","shell.execute_reply.started":"2022-08-15T01:48:31.869513Z","shell.execute_reply":"2022-08-15T01:48:31.876645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.879522Z","iopub.execute_input":"2022-08-15T01:48:31.879850Z","iopub.status.idle":"2022-08-15T01:48:31.898801Z","shell.execute_reply.started":"2022-08-15T01:48:31.879818Z","shell.execute_reply":"2022-08-15T01:48:31.897516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling the missing value of Age with nearest row\nnew_test_data = test_data.interpolate(method='nearest', axis=0)\nnew_test_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.902212Z","iopub.execute_input":"2022-08-15T01:48:31.902714Z","iopub.status.idle":"2022-08-15T01:48:31.922996Z","shell.execute_reply.started":"2022-08-15T01:48:31.902668Z","shell.execute_reply":"2022-08-15T01:48:31.921664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We divide the train dataset into data and label where X is data and Y is label\ntrainY = new_train_data['Survived'].copy()\ntrainX = new_train_data.drop(['Survived'], axis=1).copy()\ntrainX","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.924677Z","iopub.execute_input":"2022-08-15T01:48:31.925140Z","iopub.status.idle":"2022-08-15T01:48:31.948332Z","shell.execute_reply.started":"2022-08-15T01:48:31.925092Z","shell.execute_reply":"2022-08-15T01:48:31.947369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We use the same approach for test set. However, we have 2 Age rows which are not filled by nearest row. This can be filled by using median of column\n\nnew_test_data['Age'] = new_test_data['Age'].fillna(new_test_data['Age'].median())\ntestY = new_test_data['Survived'].copy()\ntestX = new_test_data.drop(['Survived'], axis=1).copy()\n\ntestX","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.949917Z","iopub.execute_input":"2022-08-15T01:48:31.950570Z","iopub.status.idle":"2022-08-15T01:48:31.978763Z","shell.execute_reply.started":"2022-08-15T01:48:31.950531Z","shell.execute_reply":"2022-08-15T01:48:31.977810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**In this step we still see that the range of each feature is imbalance which suggest us to scale data to the same range [0:1] using MinMaxScaler**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\ntrain_X = scaler.fit_transform(trainX)\ntest_X = scaler.fit_transform(testX)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.981642Z","iopub.execute_input":"2022-08-15T01:48:31.982783Z","iopub.status.idle":"2022-08-15T01:48:31.995946Z","shell.execute_reply.started":"2022-08-15T01:48:31.982722Z","shell.execute_reply":"2022-08-15T01:48:31.994879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**To make classification model, I tend to use one of most popular approach which is Logistic Regression. If you come to this step, you can choose any classification algorithm you like.**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\n\nclf = LogisticRegression().fit(train_X, trainY)\ny_pred = clf.predict(test_X)\naccuracy_score(testY, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:31.997224Z","iopub.execute_input":"2022-08-15T01:48:31.998445Z","iopub.status.idle":"2022-08-15T01:48:32.028494Z","shell.execute_reply.started":"2022-08-15T01:48:31.998406Z","shell.execute_reply":"2022-08-15T01:48:32.027085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**In the last step for machine learning project, we need to fine tune model to achieve best combination of hyper-parameters. Indeed, I ignore a validation step in this small project ^^**","metadata":{}},{"cell_type":"code","source":"# Fine Tune model\nfrom sklearn.model_selection import GridSearchCV\n\n# Fine tune parameters\nparameters = {'penalty':('l1', 'l2', 'elasticnet', 'none'),\n             'solver': ('newton-cg', 'lbfgs', 'liblinear', 'sag', 'saga'),\n             'warm_start': ('False', 'True')}\nlr = LogisticRegression()\n\n# Get the best model with fine tuning process\nclf = GridSearchCV(lr, parameters)\nclf.fit(train_X, trainY)\ny_pred = clf.predict(test_X)\naccuracy_score(testY, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:32.030057Z","iopub.execute_input":"2022-08-15T01:48:32.030503Z","iopub.status.idle":"2022-08-15T01:48:32.959067Z","shell.execute_reply.started":"2022-08-15T01:48:32.030469Z","shell.execute_reply":"2022-08-15T01:48:32.957956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The best model achieve ~96.4% of Accuracy**","metadata":{}},{"cell_type":"code","source":"# Now we can store\nresults = test_data['Survived'].copy()\nresults.values[:] = 0\nresults.values[:] = y_pred\npassId = passengerId\ndf = passId.to_frame()\ndf['Survived'] = results\ndf.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T01:48:32.960239Z","iopub.execute_input":"2022-08-15T01:48:32.960614Z","iopub.status.idle":"2022-08-15T01:48:32.970323Z","shell.execute_reply.started":"2022-08-15T01:48:32.960583Z","shell.execute_reply":"2022-08-15T01:48:32.968966Z"},"trusted":true},"execution_count":null,"outputs":[]}]}