{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T23:14:04.306845Z","iopub.execute_input":"2022-07-15T23:14:04.307299Z","iopub.status.idle":"2022-07-15T23:14:04.343664Z","shell.execute_reply.started":"2022-07-15T23:14:04.307208Z","shell.execute_reply":"2022-07-15T23:14:04.342366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Goal: to build out pipeline with GridsearchCV & Accuracy > 79.9, then submit\n# Ref 1: https://towardsdatascience.com/advanced-pipelines-with-scikit-learn-4204bb71019b    ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:18:10.898361Z","iopub.execute_input":"2022-07-15T23:18:10.898734Z","iopub.status.idle":"2022-07-15T23:18:10.903132Z","shell.execute_reply.started":"2022-07-15T23:18:10.898705Z","shell.execute_reply":"2022-07-15T23:18:10.901994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All credit to... https://towardsdatascience.com/advanced-pipelines-with-scikit-learn-4204bb71019b\n# The usual suspects\n!pip install feature_engine\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n#Sklearn\nfrom sklearn.model_selection import (train_test_split, RandomizedSearchCV, \n                                     RepeatedStratifiedKFold, cross_validate)\n\n# Assemble pipeline(s)\nfrom sklearn import set_config\nfrom sklearn.pipeline import make_pipeline, Pipeline\n# *from imblearn.pipeline import Pipeline as imbPipeline\nfrom sklearn.compose import ColumnTransformer, make_column_selector\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder, MinMaxScaler\n\n# Handle constant/duplicates and missing features/columns\nfrom feature_engine.selection import (DropFeatures, DropConstantFeatures, \n                                      DropDuplicateFeatures)\n\n# Models\nfrom xgboost import XGBClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier, VotingClassifier\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.inspection import permutation_importance\nfrom scipy.stats import loguniform\n\nset_config(display=\"diagram\")  # make pipeline visible","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:14:07.499411Z","iopub.execute_input":"2022-07-15T23:14:07.500106Z","iopub.status.idle":"2022-07-15T23:14:22.576714Z","shell.execute_reply.started":"2022-07-15T23:14:07.500056Z","shell.execute_reply":"2022-07-15T23:14:22.575577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv')\ntrain = train_data.copy()\ntest_data = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')\ntest = test_data.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:14:28.953317Z","iopub.execute_input":"2022-07-15T23:14:28.953732Z","iopub.status.idle":"2022-07-15T23:14:29.046040Z","shell.execute_reply.started":"2022-07-15T23:14:28.953701Z","shell.execute_reply":"2022-07-15T23:14:29.044735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nXX = train_data.copy()\n\nXX[['Cabin1', 'Cabin2', 'Cabin3']] = XX['Cabin'].str.split('/', expand=True)\nXX['Cabin2'] = XX['Cabin2'].astype(float)\n\nXX = XX.drop(['PassengerId', 'Name','Cabin'],axis=1)\n\ns = XX.pop('Transported')\nnew_df = pd.concat([XX, s], 1)\nXX = new_df\n\n#XX[XX.select_dtypes(['object']).columns] = XX.select_dtypes(['object']).apply(lambda x: x.astype('category'))\n#XX.dtypes\nX = XX.drop(\"Transported\", axis=1)\ny = XX[\"Transported\"]\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.25,random_state=42)\n#X_train, X_test, Y_train, y_test","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:14:31.384748Z","iopub.execute_input":"2022-07-15T23:14:31.385446Z","iopub.status.idle":"2022-07-15T23:14:31.444065Z","shell.execute_reply.started":"2022-07-15T23:14:31.385405Z","shell.execute_reply":"2022-07-15T23:14:31.443136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Linear model (logistic regression)\nlr = LogisticRegression(warm_start=True, max_iter=400)\n# RandomForest\nrf = RandomForestClassifier(random_state=42)\n# XGB\nxgb = XGBClassifier(tree_method=\"hist\", verbosity=0, silent=True)\n# Ensemble\nlr_xgb_rf = VotingClassifier(estimators=[('lr', lr), ('xgb', xgb), ('rf', rf)], \n                             voting='soft')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:14:34.822586Z","iopub.execute_input":"2022-07-15T23:14:34.822981Z","iopub.status.idle":"2022-07-15T23:14:34.830246Z","shell.execute_reply.started":"2022-07-15T23:14:34.822948Z","shell.execute_reply":"2022-07-15T23:14:34.829178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ppl = Pipeline([\n    # Step 1: Drop irrelevant columns/features\n    ('drop_constant_values', DropConstantFeatures(tol=1, missing_values='ignore')),\n    ('drop_duplicates', DropDuplicateFeatures()),\n    \n    # Step 2: Impute and scale columns/features\n    ('cleaning', ColumnTransformer([\n        # Step 2.1: Apply steps for numerical features\n        ('num',make_pipeline(\n            SimpleImputer(strategy='most_frequent'),\n        ),\n         #make_column_selector(dtype_include='int64')\n         make_column_selector(dtype_include=['int64','float64'])\n        ),\n        # Step 2.2 Apply steps for categorial features\n        ('cat',make_pipeline(\n            SimpleImputer(strategy='most_frequent'),\n            #OneHotEncoder()\n            OneHotEncoder(sparse=False, handle_unknown='ignore')\n        ),\n         make_column_selector(dtype_include=['category','object'])\n        )])\n    ),\n    \n    # Step 4: Voting Classifier\n    ('ensemble', lr_xgb_rf)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:14:37.623489Z","iopub.execute_input":"2022-07-15T23:14:37.623878Z","iopub.status.idle":"2022-07-15T23:14:37.634378Z","shell.execute_reply.started":"2022-07-15T23:14:37.623849Z","shell.execute_reply":"2022-07-15T23:14:37.633081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ppl.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:14:41.274409Z","iopub.execute_input":"2022-07-15T23:14:41.274782Z","iopub.status.idle":"2022-07-15T23:14:43.490117Z","shell.execute_reply.started":"2022-07-15T23:14:41.274753Z","shell.execute_reply":"2022-07-15T23:14:43.488714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = ppl.predict(X_val)\nfrom sklearn.metrics import accuracy_score, confusion_matrix, precision_score, recall_score, roc_auc_score, roc_curve, f1_score\naccuracy_score(y_val, y_pred)\nprint('The accuracy of the model is :', round(accuracy_score(y_val,y_pred),2)*100)\nprint('The precision of the model is:', round(precision_score(y_val,y_pred),2)*100)\nprint('The recall of the model is   :', round(recall_score(y_val,y_pred),2)*100)\nprint('The f1 score of the model is :', round(f1_score(y_val,y_pred),2)*100)\nprint('Confusion matrix:')\nprint(confusion_matrix(y_val, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:14:48.671610Z","iopub.execute_input":"2022-07-15T23:14:48.672018Z","iopub.status.idle":"2022-07-15T23:14:48.849370Z","shell.execute_reply.started":"2022-07-15T23:14:48.671987Z","shell.execute_reply":"2022-07-15T23:14:48.848130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Hyperparameter Tuning\n# params = {\n#     'ensemble__lr__solver': ['newton-cg', 'lbfgs', 'liblinear'],\n#     'ensemble__lr__penalty': ['none', 'l1', 'l2', 'elasticnet'],\n#     'ensemble__lr__C': loguniform(1e-5, 100),\n#     'ensemble__xgb__learning_rate': [0.1],\n#     'ensemble__xgb__max_depth': [7, 10, 15],\n#     'ensemble__xgb__min_child_weight': [10, 15, 20],\n#     'ensemble__xgb__colsample_bytree': [0.8, 0.9, 1],\n#     'ensemble__xgb__n_estimators': [300, 400],\n#     'ensemble__xgb__reg_alpha': [0.5, 0.2, 1],\n#     'ensemble__xgb__reg_lambda': [2, 3, 5],\n#     'ensemble__xgb__gamma': [1, 2, 3],\n#    'ensemble__rf__max_depth': [5, 10],\n#    'ensemble__rf__min_samples_leaf': [1, 2, 4],\n#    'ensemble__rf__min_samples_split': [3, 7, 10],\n#     'ensemble__rf__n_estimators': [10, 50],\n# }\n\n# rsf = RepeatedStratifiedKFold(random_state=42)\n# clf = RandomizedSearchCV(ppl, params, verbose=2, cv=rsf)\n# clf.fit(X_train, y_train)\n\n# print(\"Best Score: \", clf.best_score_)\n# print(\"Best Params: \", clf.best_params_)\n# print(\"AUC:\", roc_auc_score(y_val, clf.predict(X_val)))\n\n# y_pred = clf.predict(X_val)\n# from sklearn.metrics import accuracy_score, confusion_matrix, precision_score, recall_score, roc_auc_score, roc_curve, f1_score\n# accuracy_score(y_val, y_pred)\n# print('The accuracy of the model is', round(accuracy_score(y_val,y_pred),2)*100)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://inria.github.io/scikit-learn-mooc/python_scripts/dev_features_importance.html\ndef plot_feature_importances(perm_importance_result, feat_name):\n    \"\"\" bar plot the feature importance \"\"\"\n    fig, ax = plt.subplots()\n\n    indices = perm_importance_result['importances_mean'].argsort()\n    plt.barh(range(len(indices)),\n             perm_importance_result['importances_mean'][indices],\n             xerr=perm_importance_result['importances_std'][indices])\n    ax.set_yticks(range(len(indices)))\n    ax.set_title(\"Permutation importance\")\n    \n    tmp = np.array(feat_name)\n    _ = ax.set_yticklabels(tmp[indices])\n\n# Extract feature names after the transformation steps\n# Therefore, we have to fit one part ([0:4]) of our pipeline to our data\nppl_fts = ppl[0:1]\nppl_fts.fit(X_train, y_train)\nfeatures = ppl_fts.get_feature_names_out()\n\n# We provide the function our hyperparameter-tuned model/pipeline: clf\n# In case we do not use hyperparameter tuning, we could provide here a fitted version of ppl\n# For example: ppl.fit(X_train, y_train)\nperm_importance_result_train = permutation_importance(ppl, X_train, y_train, random_state=42)\nplot_feature_importances(perm_importance_result_train, features)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:16:14.072057Z","iopub.execute_input":"2022-07-15T23:16:14.072582Z","iopub.status.idle":"2022-07-15T23:16:28.714979Z","shell.execute_reply.started":"2022-07-15T23:16:14.072481Z","shell.execute_reply":"2022-07-15T23:16:28.713505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:16:33.448937Z","iopub.execute_input":"2022-07-15T23:16:33.449346Z","iopub.status.idle":"2022-07-15T23:16:33.453934Z","shell.execute_reply.started":"2022-07-15T23:16:33.449315Z","shell.execute_reply":"2022-07-15T23:16:33.453096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nXX = test.copy()\n\nXX[['Cabin1', 'Cabin2', 'Cabin3']] = XX['Cabin'].str.split('/', expand=True)\nXX['Cabin2'] = XX['Cabin2'].astype(float)\n\nXX = XX.drop(['PassengerId', 'Name','Cabin'],axis=1)\n\ntest_pred = XX","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:23:46.400768Z","iopub.execute_input":"2022-07-15T23:23:46.401179Z","iopub.status.idle":"2022-07-15T23:23:46.427207Z","shell.execute_reply.started":"2022-07-15T23:23:46.401145Z","shell.execute_reply":"2022-07-15T23:23:46.426050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = ppl.predict(test_pred)\ntest_pred['Transported'] = pred\ntest_pred['PassengerId'] = test['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:23:48.331515Z","iopub.execute_input":"2022-07-15T23:23:48.331929Z","iopub.status.idle":"2022-07-15T23:23:48.502443Z","shell.execute_reply.started":"2022-07-15T23:23:48.331894Z","shell.execute_reply":"2022-07-15T23:23:48.501295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = test_pred[['PassengerId', 'Transported']]\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:24:07.807438Z","iopub.execute_input":"2022-07-15T23:24:07.807835Z","iopub.status.idle":"2022-07-15T23:24:07.819863Z","shell.execute_reply.started":"2022-07-15T23:24:07.807802Z","shell.execute_reply":"2022-07-15T23:24:07.819041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\nprint('Submission succesful!')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:24:12.777568Z","iopub.execute_input":"2022-07-15T23:24:12.777987Z","iopub.status.idle":"2022-07-15T23:24:12.793918Z","shell.execute_reply.started":"2022-07-15T23:24:12.777954Z","shell.execute_reply":"2022-07-15T23:24:12.793055Z"},"trusted":true},"execution_count":null,"outputs":[]}]}