{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T00:08:42.770063Z","iopub.execute_input":"2022-07-27T00:08:42.770484Z","iopub.status.idle":"2022-07-27T00:08:42.781987Z","shell.execute_reply.started":"2022-07-27T00:08:42.770446Z","shell.execute_reply":"2022-07-27T00:08:42.780516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Import libraries and read the data","metadata":{}},{"cell_type":"code","source":"import random\nimport pandas as pd \nimport numpy as np\nimport math as ma\nimport tensorflow as tf\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import confusion_matrix,recall_score\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.preprocessing import LabelEncoder\n\n#Read the Data\ndf=pd.read_csv('../input/titanic/train.csv')\ndf_test=pd.read_csv('../input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:19:51.560385Z","iopub.execute_input":"2022-07-27T00:19:51.560864Z","iopub.status.idle":"2022-07-27T00:19:51.582045Z","shell.execute_reply.started":"2022-07-27T00:19:51.560803Z","shell.execute_reply":"2022-07-27T00:19:51.581191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Do some analysis on data\n * show the number of null samples in train , test datasets.\n * we will notice here that columns (age,cabin) have a great number of null values so we will    drop it.\n ","metadata":{}},{"cell_type":"code","source":"print(df.isna().sum())\nprint(df_test.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:11:45.580072Z","iopub.execute_input":"2022-07-27T00:11:45.580572Z","iopub.status.idle":"2022-07-27T00:11:45.597102Z","shell.execute_reply.started":"2022-07-27T00:11:45.580534Z","shell.execute_reply":"2022-07-27T00:11:45.595538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Describe the data and show some statstics.","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:11:52.057009Z","iopub.execute_input":"2022-07-27T00:11:52.057439Z","iopub.status.idle":"2022-07-27T00:11:52.093710Z","shell.execute_reply.started":"2022-07-27T00:11:52.057395Z","shell.execute_reply":"2022-07-27T00:11:52.092900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Do some preprocessing on train and test datasets\n* drop unimportant columns.\n* fill null values in embarked column with random choice.\n* divide my data to trainning features and label.\n* scale my data for easy trainig and convergence.","metadata":{}},{"cell_type":"code","source":"#preprocessing on main df \ndf.drop(columns=['Cabin','Age', 'PassengerId','Name','Ticket','Parch','SibSp'],inplace=True)\n\n\n\ndf[\"Embarked\"].fillna( random.choice(df[df['Embarked'] != np.nan][\"Embarked\"]), inplace =True)\n\n\nfrom sklearn.preprocessing import LabelEncoder\ndf=df.apply(LabelEncoder().fit_transform)\n\n\n#divide data to training and label\nX=df.iloc[ : ,1:].values\ny=df.iloc[:,0]\n\n\n\nsc = StandardScaler()\nX = sc.fit_transform(X)\nprint(X)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:02.243517Z","iopub.execute_input":"2022-07-27T00:20:02.243961Z","iopub.status.idle":"2022-07-27T00:20:02.264312Z","shell.execute_reply.started":"2022-07-27T00:20:02.243927Z","shell.execute_reply":"2022-07-27T00:20:02.262635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Repeat the same steps on testset.","metadata":{}},{"cell_type":"code","source":"#preprocessing on test df \n\ndf_test.drop(columns=['Cabin','Age', 'PassengerId','Name','Ticket','Parch','SibSp'],inplace=True)\n\n\ndf_test['Fare'].fillna((df_test['Fare'].mean()), inplace=True)\ndf_test[\"Embarked\"].fillna( random.choice(df_test[df_test['Embarked'] != np.nan][\"Embarked\"]), inplace =True)\n\n\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:08.076237Z","iopub.execute_input":"2022-07-27T00:20:08.076686Z","iopub.status.idle":"2022-07-27T00:20:08.088913Z","shell.execute_reply.started":"2022-07-27T00:20:08.076650Z","shell.execute_reply":"2022-07-27T00:20:08.087327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\n\nfrom sklearn.preprocessing import LabelEncoder\ndf_test=df_test.apply(LabelEncoder().fit_transform)\n\n# Feature Scaling\nsc = StandardScaler()\ndf_test = sc.fit_transform(df_test)\nprint(df_test)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:11.425815Z","iopub.execute_input":"2022-07-27T00:20:11.426259Z","iopub.status.idle":"2022-07-27T00:20:11.442415Z","shell.execute_reply.started":"2022-07-27T00:20:11.426225Z","shell.execute_reply":"2022-07-27T00:20:11.440916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do some visulization on tra","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nsns.pairplot(df)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:08:43.069226Z","iopub.execute_input":"2022-07-27T00:08:43.069648Z","iopub.status.idle":"2022-07-27T00:08:45.579226Z","shell.execute_reply.started":"2022-07-27T00:08:43.069615Z","shell.execute_reply":"2022-07-27T00:08:45.578039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxplot = df.boxplot(column=[ 'Fare']) ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:08:45.581326Z","iopub.execute_input":"2022-07-27T00:08:45.581703Z","iopub.status.idle":"2022-07-27T00:08:45.754645Z","shell.execute_reply.started":"2022-07-27T00:08:45.581671Z","shell.execute_reply":"2022-07-27T00:08:45.753129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:18.227709Z","iopub.execute_input":"2022-07-27T00:20:18.228211Z","iopub.status.idle":"2022-07-27T00:20:18.235875Z","shell.execute_reply.started":"2022-07-27T00:20:18.228174Z","shell.execute_reply":"2022-07-27T00:20:18.234782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import BaggingClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\n\nbagging = BaggingClassifier(AdaBoostClassifier(n_estimators=3),max_samples=0.5, max_features=0.5)\nbagging.fit(X_train,y_train)\ny_pred = bagging.predict(X_test)\n# Making the Confusion Matrix\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:29:06.994354Z","iopub.execute_input":"2022-07-27T00:29:06.994748Z","iopub.status.idle":"2022-07-27T00:29:07.091426Z","shell.execute_reply.started":"2022-07-27T00:29:06.994717Z","shell.execute_reply":"2022-07-27T00:29:07.090238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score,recall_score\nrecall=recall_score(y_pred,y_test)\nprecision=precision_score(y_pred,y_test)\nprint('recall is '+ str(recall))\nprint('precision is '+ str(precision))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:29.189796Z","iopub.execute_input":"2022-07-27T00:20:29.190187Z","iopub.status.idle":"2022-07-27T00:20:29.201767Z","shell.execute_reply.started":"2022-07-27T00:20:29.190157Z","shell.execute_reply":"2022-07-27T00:20:29.200312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:32.923941Z","iopub.execute_input":"2022-07-27T00:20:32.924298Z","iopub.status.idle":"2022-07-27T00:20:32.931876Z","shell.execute_reply.started":"2022-07-27T00:20:32.924271Z","shell.execute_reply":"2022-07-27T00:20:32.930816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final=bagging.predict(df_test)\nfinal","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:37.554612Z","iopub.execute_input":"2022-07-27T00:20:37.555849Z","iopub.status.idle":"2022-07-27T00:20:37.575735Z","shell.execute_reply.started":"2022-07-27T00:20:37.555791Z","shell.execute_reply":"2022-07-27T00:20:37.574612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(r\"../input/titanic/test.csv\")\nsubmission = pd.DataFrame({\n    \"PassengerId\" : test[\"PassengerId\"],\n    \"Survived\" : final\n})\nsubmission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:41.099456Z","iopub.execute_input":"2022-07-27T00:20:41.099923Z","iopub.status.idle":"2022-07-27T00:20:41.118584Z","shell.execute_reply.started":"2022-07-27T00:20:41.099885Z","shell.execute_reply":"2022-07-27T00:20:41.117240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:20:45.962391Z","iopub.execute_input":"2022-07-27T00:20:45.962869Z","iopub.status.idle":"2022-07-27T00:20:45.971460Z","shell.execute_reply.started":"2022-07-27T00:20:45.962801Z","shell.execute_reply":"2022-07-27T00:20:45.970395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-27T00:15:47.412018Z","iopub.execute_input":"2022-07-27T00:15:47.412431Z","iopub.status.idle":"2022-07-27T00:15:47.431684Z","shell.execute_reply.started":"2022-07-27T00:15:47.412401Z","shell.execute_reply":"2022-07-27T00:15:47.430372Z"},"trusted":true},"execution_count":null,"outputs":[]}]}