{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T08:42:18.113498Z","iopub.execute_input":"2022-08-01T08:42:18.113972Z","iopub.status.idle":"2022-08-01T08:42:18.127948Z","shell.execute_reply.started":"2022-08-01T08:42:18.113931Z","shell.execute_reply":"2022-08-01T08:42:18.126435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set()\nfrom scipy import stats\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBRegressor\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import MinMaxScaler,StandardScaler,OrdinalEncoder,FunctionTransformer, RobustScaler, Normalizer,OneHotEncoder\nfrom sklearn.metrics import mean_absolute_error\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.decomposition import PCA\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVR\nfrom sklearn import svm","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:18.158875Z","iopub.execute_input":"2022-08-01T08:42:18.160105Z","iopub.status.idle":"2022-08-01T08:42:18.170855Z","shell.execute_reply.started":"2022-08-01T08:42:18.160030Z","shell.execute_reply":"2022-08-01T08:42:18.169862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_train_data='../input/titanic/train.csv'\ninput_test_data='../input/titanic/test.csv'\ntrain_data=pd.read_csv(input_train_data,index_col=0)\ntest_data=pd.read_csv(input_test_data,index_col=0)\npd.DataFrame(train_data.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:18.227129Z","iopub.execute_input":"2022-08-01T08:42:18.227763Z","iopub.status.idle":"2022-08-01T08:42:18.268996Z","shell.execute_reply.started":"2022-08-01T08:42:18.227727Z","shell.execute_reply":"2022-08-01T08:42:18.268102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y = train_data.Survived\n# X = train_data.drop(['Survived'], axis=1)\n# pd.DataFrame(X.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:18.270547Z","iopub.execute_input":"2022-08-01T08:42:18.270860Z","iopub.status.idle":"2022-08-01T08:42:18.274984Z","shell.execute_reply.started":"2022-08-01T08:42:18.270829Z","shell.execute_reply":"2022-08-01T08:42:18.274095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:18.312269Z","iopub.execute_input":"2022-08-01T08:42:18.313584Z","iopub.status.idle":"2022-08-01T08:42:18.367838Z","shell.execute_reply.started":"2022-08-01T08:42:18.313489Z","shell.execute_reply":"2022-08-01T08:42:18.366806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#catagorical colums\ncat_columns=[cname for cname in train_data.columns\n             if   train_data[cname].nunique() <= 15 and train_data[cname].dtype == \"object\"]\nprint(cat_columns)\n\nnum_columns = [nname for nname in train_data.columns if train_data[nname].dtype in ['int64', 'float64']]\n\nmy_features=cat_columns+num_columns\n\n\nnum_columns.remove('Survived')\nprint(num_columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:18.369545Z","iopub.execute_input":"2022-08-01T08:42:18.369819Z","iopub.status.idle":"2022-08-01T08:42:18.382157Z","shell.execute_reply.started":"2022-08-01T08:42:18.369785Z","shell.execute_reply":"2022-08-01T08:42:18.380540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#catagorical colums\ncat_columns_test=[cname for cname in test_data.columns\n             if   test_data[cname].nunique() <= 15 and test_data[cname].dtype == \"object\"]\nprint(cat_columns)\n\nnum_columns_test = [nname for nname in test_data.columns if test_data[nname].dtype in ['int64', 'float64']]\n\nmy_features_test=cat_columns_test+num_columns_test\n\n\n# num_columns_test.remove('Survived')\nprint(num_columns_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:18.393713Z","iopub.execute_input":"2022-08-01T08:42:18.394665Z","iopub.status.idle":"2022-08-01T08:42:18.410613Z","shell.execute_reply.started":"2022-08-01T08:42:18.394614Z","shell.execute_reply":"2022-08-01T08:42:18.409392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Correlations\ncorrelations = train_data[my_features].corr()\nf, ax = plt.subplots(figsize=(20, 15))\nsns.heatmap(correlations, square=True, cbar=True, annot=True, vmax=.9);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:18.439961Z","iopub.execute_input":"2022-08-01T08:42:18.440325Z","iopub.status.idle":"2022-08-01T08:42:19.041818Z","shell.execute_reply.started":"2022-08-01T08:42:18.440290Z","shell.execute_reply":"2022-08-01T08:42:19.040693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data[num_columns].hist(figsize=(24,12))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:19.044842Z","iopub.execute_input":"2022-08-01T08:42:19.046096Z","iopub.status.idle":"2022-08-01T08:42:19.051610Z","shell.execute_reply.started":"2022-08-01T08:42:19.046033Z","shell.execute_reply":"2022-08-01T08:42:19.050306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"length_before=len(train_data)\nlength_after=len(train_data[my_features].drop_duplicates())\ntrain_data=train_data[my_features].drop_duplicates()\nprint(length_before,\"===>\",length_after)\nprint(len(train_data))\n\n\n\nlength_before_1=len(test_data)\nlength_after_1=len(test_data[my_features_test].drop_duplicates())\n# test_data=test_data[my_features_test].drop_duplicates()\n# print(length_before_1,\"===>\",length_after_1)\n# print(len(test_data))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:19.053147Z","iopub.execute_input":"2022-08-01T08:42:19.053925Z","iopub.status.idle":"2022-08-01T08:42:19.084135Z","shell.execute_reply.started":"2022-08-01T08:42:19.053869Z","shell.execute_reply":"2022-08-01T08:42:19.083367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_with_missing_train = [col for col in train_data.columns\n                     if train_data[col].isnull().any()]\n                     \ncols_with_missing_test = [col for col in test_data.columns\n                     if test_data[col].isnull().any()]\n\nprint(cols_with_missing_train,\"===>\", cols_with_missing_test)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:19.086460Z","iopub.execute_input":"2022-08-01T08:42:19.087116Z","iopub.status.idle":"2022-08-01T08:42:19.100071Z","shell.execute_reply.started":"2022-08-01T08:42:19.087080Z","shell.execute_reply":"2022-08-01T08:42:19.099021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(18,6))\nsns.boxplot(data=train_data[num_columns], orient=\"h\", palette=\"Set2\");\nplt.xticks(fontsize= 14)\nplt.title('Box plot of numerical columns', fontsize=16);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:19.101984Z","iopub.execute_input":"2022-08-01T08:42:19.102538Z","iopub.status.idle":"2022-08-01T08:42:19.477388Z","shell.execute_reply.started":"2022-08-01T08:42:19.102364Z","shell.execute_reply":"2022-08-01T08:42:19.476337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(18,6))\nsns.boxplot(data=test_data[num_columns_test], orient=\"h\", palette=\"Set2\");\nplt.xticks(fontsize= 14)\nplt.title('Box plot of numerical columns', fontsize=16);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:19.478889Z","iopub.execute_input":"2022-08-01T08:42:19.479491Z","iopub.status.idle":"2022-08-01T08:42:19.881583Z","shell.execute_reply.started":"2022-08-01T08:42:19.479446Z","shell.execute_reply":"2022-08-01T08:42:19.880637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(data=train_data[['Survived']], orient=\"h\", palette=\"Set2\" );\nplt.xticks(fontsize= 14)\nplt.title('Box plot of target column', fontsize=16);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:19.882991Z","iopub.execute_input":"2022-08-01T08:42:19.883290Z","iopub.status.idle":"2022-08-01T08:42:20.187076Z","shell.execute_reply.started":"2022-08-01T08:42:19.883256Z","shell.execute_reply":"2022-08-01T08:42:20.186015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def treatoutliers(df=None, columns=None, factor=1.5, method='IQR', treatment='cap'):\n\n    for column in columns:\n        if method == 'STD':\n            permissable_std = factor * df[column].std()\n            col_mean = df[column].mean()\n            floor, ceil = col_mean - permissable_std, col_mean + permissable_std\n        elif method == 'IQR':\n            Q1 = df[column].quantile(0.25)\n            Q3 = df[column].quantile(0.75)\n            IQR = Q3 - Q1\n            floor, ceil = Q1 - factor * IQR, Q3 + factor * IQR\n#         print(floor, ceil)\n        if treatment == 'remove':\n            print(treatment, column)\n            df = df[(df[column] >= floor) & (df[column] <= ceil)]\n            # link for   https://www.geeksforgeeks.org/numpy-clip-in-python/\n            # clip to make all the data between the q1 and q3 and not make the data in outliers \n        elif treatment == 'cap':\n            print(treatment, column)\n            df[column] = df[column].clip(floor, ceil)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:20.188777Z","iopub.execute_input":"2022-08-01T08:42:20.189147Z","iopub.status.idle":"2022-08-01T08:42:20.202193Z","shell.execute_reply.started":"2022-08-01T08:42:20.189100Z","shell.execute_reply":"2022-08-01T08:42:20.200762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for colName in [['Age', 'SibSp', 'Parch', 'Fare']]:\n    train_data = treatoutliers(df=train_data,columns=colName, treatment='cap')      \n    \ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:20.204078Z","iopub.execute_input":"2022-08-01T08:42:20.204755Z","iopub.status.idle":"2022-08-01T08:42:20.255863Z","shell.execute_reply.started":"2022-08-01T08:42:20.204702Z","shell.execute_reply":"2022-08-01T08:42:20.254993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for colName in [['Age', 'SibSp', 'Parch', 'Fare']]:\n    test_data = treatoutliers(df=test_data,columns=colName, treatment='cap')      \n    \ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:20.258809Z","iopub.execute_input":"2022-08-01T08:42:20.259627Z","iopub.status.idle":"2022-08-01T08:42:20.304034Z","shell.execute_reply.started":"2022-08-01T08:42:20.259589Z","shell.execute_reply":"2022-08-01T08:42:20.302748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(18,6))\nsns.boxplot(data=train_data[num_columns], orient=\"h\", palette=\"Set2\");\nplt.xticks(fontsize= 14)\nplt.title('Box plot of numerical columns', fontsize=16);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:20.305675Z","iopub.execute_input":"2022-08-01T08:42:20.305970Z","iopub.status.idle":"2022-08-01T08:42:20.746228Z","shell.execute_reply.started":"2022-08-01T08:42:20.305937Z","shell.execute_reply":"2022-08-01T08:42:20.745021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(18,6))\nsns.boxplot(data=test_data[num_columns_test], orient=\"h\", palette=\"Set2\");\nplt.xticks(fontsize= 14)\nplt.title('Box plot of numerical columns', fontsize=16);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:20.747796Z","iopub.execute_input":"2022-08-01T08:42:20.748140Z","iopub.status.idle":"2022-08-01T08:42:21.163426Z","shell.execute_reply.started":"2022-08-01T08:42:20.748106Z","shell.execute_reply":"2022-08-01T08:42:21.161973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_data.Survived\nX = train_data.drop(['Survived'], axis=1)\npd.DataFrame(X.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.165325Z","iopub.execute_input":"2022-08-01T08:42:21.166183Z","iopub.status.idle":"2022-08-01T08:42:21.187374Z","shell.execute_reply.started":"2022-08-01T08:42:21.166126Z","shell.execute_reply":"2022-08-01T08:42:21.186403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"women = train_data.loc[train_data.Sex == 'female'][\"Survived\"]\nrate_women = sum(women)/len(women)\n\nprint(\"% of women who survived:\", rate_women) \n\nmen = train_data.loc[train_data.Sex == 'male'][\"Survived\"]\nrate_men = sum(men)/len(men)\n\nprint(\"% of men who survived:\", rate_men)\n\n\nagegreater15 = train_data.loc[train_data.Age>15][\"Survived\"]\nrate_agegreater15 = sum(agegreater15)/len(agegreater15)\n\nprint(\"% of rate_agegreater15:\", rate_agegreater15)\n\nageless15 = train_data.loc[train_data.Age<=15][\"Survived\"]\nrate_ageless15 = sum(ageless15)/len(ageless15)\n\nprint(\"% of ageless15:\", rate_ageless15)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.189138Z","iopub.execute_input":"2022-08-01T08:42:21.189778Z","iopub.status.idle":"2022-08-01T08:42:21.212669Z","shell.execute_reply.started":"2022-08-01T08:42:21.189738Z","shell.execute_reply":"2022-08-01T08:42:21.211246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_valid,y_train,y_valid=train_test_split(X, y, train_size=0.8, test_size=0.2, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.215342Z","iopub.execute_input":"2022-08-01T08:42:21.216187Z","iopub.status.idle":"2022-08-01T08:42:21.229542Z","shell.execute_reply.started":"2022-08-01T08:42:21.216127Z","shell.execute_reply":"2022-08-01T08:42:21.228035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rans = 42\nxgb_params = {'n_estimators': 150, 'max_depth': 3, 'learning_rate': 0.01,\n              'gamma': 0, 'min_child_weight': 1, 'subsample': 0.7875490025178415, \n              'colsample_bytree': 0.11807135201147481, 'reg_alpha': 23.13181079976304, \n              'reg_lambda': 0.0008746338866473539, 'random_state':rans}\n# model = XGBRegressor(**xgb_params) is 47 error\n# model=GaussianNB() is 53 error\n# model=LogisticRegression() is 50 error\nmodel= SVR(epsilon=0.2) # is error 43 with out epsilon     with 0.2   41  \n# model= svm.SVC() is error 48","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.231361Z","iopub.execute_input":"2022-08-01T08:42:21.232488Z","iopub.status.idle":"2022-08-01T08:42:21.245850Z","shell.execute_reply.started":"2022-08-01T08:42:21.232408Z","shell.execute_reply":"2022-08-01T08:42:21.243934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_transformer = Pipeline(steps=[\n       ('imputer', SimpleImputer(strategy='mean'))\n       #,('transformer', transformer)\n       ,('RobustScaler', RobustScaler(with_centering=True, with_scaling=True, quantile_range=(25.0, 75.0), copy=True))  \n       ,('scaler', StandardScaler()\n        )\n   \n])\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')) \n    ,('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, num_columns),\n        ('cat', categorical_transformer, cat_columns)\n    ],\n    remainder=\"passthrough\"\n  )","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.251883Z","iopub.execute_input":"2022-08-01T08:42:21.253336Z","iopub.status.idle":"2022-08-01T08:42:21.265161Z","shell.execute_reply.started":"2022-08-01T08:42:21.253257Z","shell.execute_reply":"2022-08-01T08:42:21.263679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = Pipeline(steps=[('preprocessor', preprocessor),\n                      ('model', model)\n                     ])\n\nfinal_model = clf.fit(x_train, y_train) ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.266613Z","iopub.execute_input":"2022-08-01T08:42:21.267010Z","iopub.status.idle":"2022-08-01T08:42:21.341903Z","shell.execute_reply.started":"2022-08-01T08:42:21.266970Z","shell.execute_reply":"2022-08-01T08:42:21.340963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = final_model.predict(x_valid)\nprint('RMSE:',mean_squared_error(y_valid, predictions, squared=False))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.343248Z","iopub.execute_input":"2022-08-01T08:42:21.344277Z","iopub.status.idle":"2022-08-01T08:42:21.360562Z","shell.execute_reply.started":"2022-08-01T08:42:21.344221Z","shell.execute_reply":"2022-08-01T08:42:21.359762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.index","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.361855Z","iopub.execute_input":"2022-08-01T08:42:21.362163Z","iopub.status.idle":"2022-08-01T08:42:21.376514Z","shell.execute_reply.started":"2022-08-01T08:42:21.362130Z","shell.execute_reply":"2022-08-01T08:42:21.375296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['PassengerId']=test_data.index","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.378299Z","iopub.execute_input":"2022-08-01T08:42:21.378545Z","iopub.status.idle":"2022-08-01T08:42:21.388466Z","shell.execute_reply.started":"2022-08-01T08:42:21.378515Z","shell.execute_reply":"2022-08-01T08:42:21.387285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.391217Z","iopub.execute_input":"2022-08-01T08:42:21.391770Z","iopub.status.idle":"2022-08-01T08:42:21.430124Z","shell.execute_reply.started":"2022-08-01T08:42:21.391722Z","shell.execute_reply":"2022-08-01T08:42:21.428961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = final_model.predict(test_data)\n# print(test_data)\nmy_submission = pd.DataFrame({'PassengerId': test_data.PassengerId, 'Survived':predictions})\nmy_submission.to_csv('gender_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:21.432716Z","iopub.execute_input":"2022-08-01T08:42:21.434141Z","iopub.status.idle":"2022-08-01T08:42:21.465076Z","shell.execute_reply.started":"2022-08-01T08:42:21.434020Z","shell.execute_reply":"2022-08-01T08:42:21.463661Z"},"trusted":true},"execution_count":null,"outputs":[]}]}