{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip  install imbalanced-learn==0.7.0\n# !pip install numba==1.20\n# !pip install --ignore-installed pycaret","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:33.231915Z","iopub.execute_input":"2022-07-22T12:34:33.232447Z","iopub.status.idle":"2022-07-22T12:34:33.257379Z","shell.execute_reply.started":"2022-07-22T12:34:33.232345Z","shell.execute_reply":"2022-07-22T12:34:33.256026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \npd.set_option('max_rows',100)\npd.set_option('max_columns',100)\nimport seaborn as sns\nimport matplotlib.pyplot as  plt\nimport scipy\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import KFold, cross_val_score\n# from pycaret.classification import setup, compare_models\nfrom sklearn.neighbors import KNeighborsRegressor\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T12:34:33.259257Z","iopub.execute_input":"2022-07-22T12:34:33.259754Z","iopub.status.idle":"2022-07-22T12:34:34.949209Z","shell.execute_reply.started":"2022-07-22T12:34:33.259723Z","shell.execute_reply":"2022-07-22T12:34:34.948317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/titanic/train.csv')\ndf_test = pd.read_csv('../input/titanic/test.csv')\nsample_sub = pd.read_csv('../input/titanic/gender_submission.csv')\n\ntarget = df_train.Survived\ndf_train.drop(['PassengerId', 'Name','Ticket', 'Survived'], axis = 1 , inplace  = True )\ndf_test.drop(['PassengerId', 'Name','Ticket'], axis = 1 , inplace  = True )\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:34.951308Z","iopub.execute_input":"2022-07-22T12:34:34.951741Z","iopub.status.idle":"2022-07-22T12:34:35.007675Z","shell.execute_reply.started":"2022-07-22T12:34:34.951697Z","shell.execute_reply":"2022-07-22T12:34:35.006737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cdata = pd.concat([df_train,df_test], axis = 0)\ncdata['Cabin'].fillna('none',inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:35.010922Z","iopub.execute_input":"2022-07-22T12:34:35.011792Z","iopub.status.idle":"2022-07-22T12:34:35.021818Z","shell.execute_reply.started":"2022-07-22T12:34:35.011748Z","shell.execute_reply":"2022-07-22T12:34:35.020559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"#ref : https://www.kaggle.com/code/ldfreeman3/a-data-science-framework-to-achieve-99-accuracy\n\ncdata['FamilySize'] = cdata['SibSp'] + cdata['Parch'] + 1\ncdata['IsAlone'] = 1\ncdata['IsAlone'].loc[cdata['FamilySize'] > 1] = 0\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:35.023587Z","iopub.execute_input":"2022-07-22T12:34:35.024428Z","iopub.status.idle":"2022-07-22T12:34:35.038288Z","shell.execute_reply.started":"2022-07-22T12:34:35.024383Z","shell.execute_reply":"2022-07-22T12:34:35.036772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ncdata2 = cdata.copy()\n\ndef knn_impute(df, na_target):\n  #imputes using knn(avg of all the nearest datapoints)  data must contain atleast one non na column\n  df = df.copy()\n\n  numeric_df = df.select_dtypes(np.number)\n  non_na_columns = numeric_df.loc[:, numeric_df.isna().sum() == 0].columns\n\n  y_train = numeric_df.loc[numeric_df[na_target].isna() == False, na_target]\n  x_train =numeric_df.loc[numeric_df[na_target].isna() == False, non_na_columns]\n  x_test = numeric_df.loc[numeric_df[na_target].isna() == True, non_na_columns]\n\n  knn = KNeighborsRegressor()\n  knn.fit(x_train, y_train)\n\n  y_pred = knn.predict(x_test)\n\n  df.loc[df[na_target].isna() == True, na_target] = y_pred\n\n  return df\n\ncdata2 = knn_impute(cdata2,'Age')\ncdata2 = knn_impute(cdata2,'Fare')\n\ncdata2['Embarked'].fillna(cdata2['Embarked'].mode()[0],inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:35.040262Z","iopub.execute_input":"2022-07-22T12:34:35.041115Z","iopub.status.idle":"2022-07-22T12:34:35.077687Z","shell.execute_reply.started":"2022-07-22T12:34:35.041058Z","shell.execute_reply":"2022-07-22T12:34:35.076843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cdata3 = cdata2.copy()\n\n\n\n\ndef skew_log_transform(df):\n  # using log1p transform to fix skew in data\n  df = df.copy()\n  numeric_features = df.select_dtypes(np.number).columns\n  for column in numeric_features:\n    skew = abs(scipy.stats.skew(df[column]))\n  \n    if skew >= 0.7:\n      df[column] = np.log1p(df[column])\n      # l1p = np.log1p(df[column])\n   \n      # tskew = abs(scipy.stats.skew(l1p))\n      # # print(tskew)\n      # # print(l1p)\n      # if tskew < skew:\n      #   df[column] = l1p\n  return df\n\ncdata3 = skew_log_transform(cdata3)\n\ncdata3 = pd.get_dummies(cdata3, drop_first = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:35.079763Z","iopub.execute_input":"2022-07-22T12:34:35.080739Z","iopub.status.idle":"2022-07-22T12:34:35.104435Z","shell.execute_reply.started":"2022-07-22T12:34:35.080692Z","shell.execute_reply":"2022-07-22T12:34:35.103532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cdata4 = cdata3.copy()\n# scaler = StandardScaler()\n# scaler.fit(cdata4)\n# cdata4 = pd.DataFrame(scaler.transform(cdata4), index = cdata4.index, columns = cdata4.columns)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:35.106272Z","iopub.execute_input":"2022-07-22T12:34:35.107032Z","iopub.status.idle":"2022-07-22T12:34:35.112849Z","shell.execute_reply.started":"2022-07-22T12:34:35.106972Z","shell.execute_reply":"2022-07-22T12:34:35.112032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = cdata4.iloc[:891,:]\nx_test = cdata4.iloc[891:,:]\n\nfinal = pd.concat([x_train, target], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:35.114208Z","iopub.execute_input":"2022-07-22T12:34:35.115277Z","iopub.status.idle":"2022-07-22T12:34:35.125345Z","shell.execute_reply.started":"2022-07-22T12:34:35.115236Z","shell.execute_reply":"2022-07-22T12:34:35.124099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# stup = setup(final,target = 'Survived')\n# best = compare_models()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:35.130402Z","iopub.execute_input":"2022-07-22T12:34:35.130754Z","iopub.status.idle":"2022-07-22T12:34:35.137365Z","shell.execute_reply.started":"2022-07-22T12:34:35.130723Z","shell.execute_reply":"2022-07-22T12:34:35.136486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import VotingClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.linear_model import RidgeClassifier, LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:35.138503Z","iopub.execute_input":"2022-07-22T12:34:35.138795Z","iopub.status.idle":"2022-07-22T12:34:36.713166Z","shell.execute_reply.started":"2022-07-22T12:34:35.138763Z","shell.execute_reply":"2022-07-22T12:34:36.712158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models  = {'catboost':CatBoostClassifier(verbose = 0,depth= 11, iterations=100, learning_rate = 0.1),\n         'gbc':GradientBoostingClassifier(),\n          'lgbm': LGBMClassifier(learning_rate = 0.1, n_estimators = 50, num_leaves = 15),\n          'xgboost' : XGBClassifier(learning_rate = 0.1, max_depth = 9, n_estimators = 50),\n           'lr' :LogisticRegression(),\n         }\n\nfor name,model in models.items():\n    model.fit(x_train, target)\n    print('---{} trained ---'.format(name))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:36.714646Z","iopub.execute_input":"2022-07-22T12:34:36.715853Z","iopub.status.idle":"2022-07-22T12:34:39.478615Z","shell.execute_reply.started":"2022-07-22T12:34:36.715804Z","shell.execute_reply":"2022-07-22T12:34:39.474486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = {}\n\nkf = KFold(n_splits = 10)\n\nfor name, model in models.items():\n    \n    result = cross_val_score(model,x_train, target, cv = kf)\n    results[name] = result\n    \nfor name, result in results.items():\n    print(\"----------\\n\" + name)\n    print(np.mean(result))\n    print(np.std(result))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:39.480829Z","iopub.execute_input":"2022-07-22T12:34:39.482118Z","iopub.status.idle":"2022-07-22T12:34:58.160786Z","shell.execute_reply.started":"2022-07-22T12:34:39.482045Z","shell.execute_reply":"2022-07-22T12:34:58.159578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----------\ncatboost\n0.8272159800249689\n0.04269662921348315\n----------\ngbc\n0.8238202247191012\n0.04131654861854713\n----------\nlgbm\n0.818214731585518\n0.036314545710137716\n----------\nxgboost\n0.811498127340824\n0.03721273000377758\n----------\nlr\n0.8035830212234705\n0.017666194624795305","metadata":{}},{"cell_type":"code","source":"estimators = [('catboost', CatBoostClassifier(verbose = 0,depth= 11, iterations=100, learning_rate = 0.1)), ('gbc', GradientBoostingClassifier()),  ('lgm', LGBMClassifier(learning_rate = 0.1, n_estimators = 50, num_leaves = 15)), ('xgboost', XGBClassifier(learning_rate = 0.1, max_depth = 9, n_estimators = 50))]\neclf = VotingClassifier(estimators=estimators, voting='soft', weights=[2,2, 1, 1])\neclf.fit(x_train,target)\ny_pred = eclf.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:34:58.162782Z","iopub.execute_input":"2022-07-22T12:34:58.163612Z","iopub.status.idle":"2022-07-22T12:35:00.178304Z","shell.execute_reply.started":"2022-07-22T12:34:58.163559Z","shell.execute_reply":"2022-07-22T12:35:00.177151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model  = CatBoostClassifier(verbose = 0,depth= 11, iterations=100, learning_rate = 0.1)\n# model.fit(x_train, target)\n# y_pred = model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:35:00.179786Z","iopub.execute_input":"2022-07-22T12:35:00.180288Z","iopub.status.idle":"2022-07-22T12:35:00.184616Z","shell.execute_reply.started":"2022-07-22T12:35:00.180254Z","shell.execute_reply":"2022-07-22T12:35:00.183860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub['Survived'] = y_pred\n\nsample_sub.to_csv('submission.csv',index= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:35:00.186448Z","iopub.execute_input":"2022-07-22T12:35:00.187838Z","iopub.status.idle":"2022-07-22T12:35:00.201172Z","shell.execute_reply.started":"2022-07-22T12:35:00.187786Z","shell.execute_reply":"2022-07-22T12:35:00.199896Z"},"trusted":true},"execution_count":null,"outputs":[]}]}