{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier, VotingClassifier, HistGradientBoostingClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OrdinalEncoder, OneHotEncoder, StandardScaler, PolynomialFeatures\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import LogisticRegression\n\nfrom sklearn.model_selection import cross_val_score, train_test_split\nfrom sklearn.metrics import classification_report\n\nfrom catboost import CatBoostClassifier, Pool\nfrom collections import Counter\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T00:12:52.519040Z","iopub.execute_input":"2022-07-21T00:12:52.520473Z","iopub.status.idle":"2022-07-21T00:12:54.385795Z","shell.execute_reply.started":"2022-07-21T00:12:52.520319Z","shell.execute_reply":"2022-07-21T00:12:54.384863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv')\ntest_data = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:12:54.388037Z","iopub.execute_input":"2022-07-21T00:12:54.389148Z","iopub.status.idle":"2022-07-21T00:12:54.480820Z","shell.execute_reply.started":"2022-07-21T00:12:54.389102Z","shell.execute_reply":"2022-07-21T00:12:54.479659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering\nGoing to do some basic feature engineering here. A lot of this is based on the summary stats I'm seeing in the data overview","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    df['pid1'] = df['PassengerId'].apply(lambda x: int(x.split(\"_\")[0]))\n    df['pid2'] = df['PassengerId'].apply(lambda x: int(x.split(\"_\")[1]))\n    df['pid1_over1k'] = df['pid1'].map(lambda x: int(x/300))\n\n    if 'Transported' in df.columns:\n        df['Transported'] = df['Transported'].map(lambda x: int(x))\n\n    df['Cabin'] = df['Cabin'].map(lambda x: str(x))\n    df['cabin1'] = df['Cabin'].apply(lambda x: x.split(\"/\")[0] if x != 'nan' else None)\n    df['cabin2'] = df['Cabin'].apply(lambda x: x.split(\"/\")[1] if x != 'nan' else None)\n    df['cabin3'] = df['Cabin'].apply(lambda x: x.split(\"/\")[2] if x != 'nan' else None)\n    \n    people_per_cabin = dict(Counter(df['Cabin']))\n    people_per_cabin.pop('nan')\n    df['cabin_mates'] = df['Cabin'].apply(lambda x: people_per_cabin[x]-1 if x in people_per_cabin else -1)\n    \n    df['Age'] = df['Age'] / 100.0\n    df['young'] = df['Age'].apply(lambda x: int(x <= 0.2)) \n    df['mid'] = df['Age'].apply(lambda x: int(0.2 < x <= 0.4 ))\n    df['old'] = df['Age'].apply(lambda x: int(0.4 < x))\n\n    for feature in ['RoomService', 'ShoppingMall', 'FoodCourt', 'Spa', 'VRDeck']:\n        df[feature] = np.log(df[feature] + 1e-2)\n        df[feature.lower()+'_lt_zero'] = df[feature].map(lambda x: int(x < -2))\n        df[feature.lower()+'_scaled'] = df[feature] / df['cabin_mates'].map(lambda x: max(x, 1))\n        \n    df['foodcourt_thresh'] = df['FoodCourt'].map(lambda x: int(x > 7))\n    df['room_thresh'] = df['RoomService'].map(lambda x: int(x > 5))\n    df['shop_thresh'] = df['ShoppingMall'].map(lambda x: int(x >= 8))\n    df['vr_thresh'] = df['VRDeck'].map(lambda x: int(x >= 8))\n\n    for feat in ['HomePlanet', 'CryoSleep', 'Destination', 'VIP']:\n        df[feat] = df[feat].fillna(\"OTHER_UNKNOWN\")\n        \n    for col in ['PassengerId','HomePlanet','CryoSleep','Cabin',\n                'Destination','VIP','Name', 'cabin1',\n                'cabin3']:\n        df[col] = df[col].map(lambda x: str(x))\n        \n    df['fname'] = df['Name'].apply(lambda x: x.split(\" \")[0] if x else \"\").fillna('')\n    df['lname'] = df['Name'].apply(lambda x: x.split(\" \")[1] if x and len(x.split(\" \")) > 1 else \"\").fillna('')\n    df['fname_len'] = df['fname'].map(lambda x: len(x))\n    df['lname_len'] = df['lname'].map(lambda x: len(x))\n    df['name_len'] = df['Name'].map(lambda x: len(x))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:12:54.482829Z","iopub.execute_input":"2022-07-21T00:12:54.483560Z","iopub.status.idle":"2022-07-21T00:12:54.509704Z","shell.execute_reply.started":"2022-07-21T00:12:54.483509Z","shell.execute_reply":"2022-07-21T00:12:54.508374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = feature_engineering(train_data)\ntest_data = feature_engineering(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:12:54.513192Z","iopub.execute_input":"2022-07-21T00:12:54.514121Z","iopub.status.idle":"2022-07-21T00:12:54.966969Z","shell.execute_reply.started":"2022-07-21T00:12:54.514072Z","shell.execute_reply":"2022-07-21T00:12:54.965536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore data\n\nPlot features, comparing their values in the non-transported vs transported groups","metadata":{}},{"cell_type":"code","source":"for feat in sorted(train_data.columns):\n    if train_data[feat].dtype not in [np.object, np.str] or len(train_data[feat].unique()) < 100:\n        plt.hist(\n            train_data[train_data['Transported'] == 0][feat],\n            alpha=0.5\n        )\n        plt.hist(\n            train_data[train_data['Transported'] == 1][feat],\n            alpha=0.5\n        )\n        plt.title(feat)\n        plt.legend([\"Not transported\", \"Transported\"])\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:12:54.968181Z","iopub.execute_input":"2022-07-21T00:12:54.968518Z","iopub.status.idle":"2022-07-21T00:13:04.254197Z","shell.execute_reply.started":"2022-07-21T00:12:54.968486Z","shell.execute_reply":"2022-07-21T00:13:04.252881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(train_data[train_data['Transported'] == 0]['pid1'].map(lambda x: int(x/300)), alpha=0.5, bins=31)\nplt.hist(train_data[train_data['Transported'] == 1]['pid1'].map(lambda x: int(x/300)), alpha=0.5, bins=31)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:13:04.257796Z","iopub.execute_input":"2022-07-21T00:13:04.258128Z","iopub.status.idle":"2022-07-21T00:13:04.563181Z","shell.execute_reply.started":"2022-07-21T00:13:04.258097Z","shell.execute_reply":"2022-07-21T00:13:04.561565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Simple sklearn data pipeline","metadata":{}},{"cell_type":"code","source":"train_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:13:04.564522Z","iopub.execute_input":"2022-07-21T00:13:04.566056Z","iopub.status.idle":"2022-07-21T00:13:04.573287Z","shell.execute_reply.started":"2022-07-21T00:13:04.566022Z","shell.execute_reply":"2022-07-21T00:13:04.572224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns = [\n    'HomePlanet',\n    'CryoSleep',\n    'Destination',\n    'VIP',\n    'cabin1',\n    'cabin3',\n    'name_len',\n    'fname_len',\n    'lname_len',\n    'cabin_mates',\n    'pid1_over1k'\n]\n\nordinal_cols = [\n    'Cabin',\n#     'pid1',\n    'cabin2'\n]\n\nnumerical_columns = [\n    'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck', \n#     'roomservice_scaled', 'foodcourt_scaled', 'shoppingmall_scaled', 'spa_scaled', 'vrdeck_scaled', \n    'Age', 'pid2', 'cabin2', 'young', \n#     'mid',\n    'roomservice_lt_zero', 'shoppingmall_lt_zero',\n    'foodcourt_lt_zero', 'spa_lt_zero', 'vrdeck_lt_zero',\n#     'room_thresh', 'vr_thresh'\n]\n\ntext_columns = ['fname', 'lname']","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:13:04.574743Z","iopub.execute_input":"2022-07-21T00:13:04.575759Z","iopub.status.idle":"2022-07-21T00:13:04.585290Z","shell.execute_reply.started":"2022-07-21T00:13:04.575714Z","shell.execute_reply":"2022-07-21T00:13:04.583543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_encoder = OneHotEncoder(\n    handle_unknown=\"ignore\"\n)\n\nordinal_encoder = Pipeline([\n    ('ordinal', OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value=-1,)),\n    ('scale', StandardScaler())\n])\n\nnumerical_pipe = Pipeline([\n    ('impute', SimpleImputer(strategy=\"mean\")), \n    ('scale', StandardScaler()),\n])\n\npreprocessing = ColumnTransformer(\n    [\n        (\"cat\", categorical_encoder, categorical_columns),\n        (\"num\", numerical_pipe, numerical_columns),\n        (\"ord\", ordinal_encoder, ordinal_cols),\n    ],\n    verbose_feature_names_out=False,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:13:04.586741Z","iopub.execute_input":"2022-07-21T00:13:04.587052Z","iopub.status.idle":"2022-07-21T00:13:04.596611Z","shell.execute_reply.started":"2022-07-21T00:13:04.587022Z","shell.execute_reply":"2022-07-21T00:13:04.595548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pp = preprocessing.fit(train_data)\ntrain_x = train_pp.transform(train_data)\ntrain_y = train_data['Transported']\n\nprint(train_x.shape)\n\n# I like plotting the feature matrix, just to eyeball it\n# plt.pcolormesh(train_x)\n# plt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:13:04.600621Z","iopub.execute_input":"2022-07-21T00:13:04.601001Z","iopub.status.idle":"2022-07-21T00:13:04.849113Z","shell.execute_reply.started":"2022-07-21T00:13:04.600969Z","shell.execute_reply":"2022-07-21T00:13:04.847822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x, val_x, train_y, val_y = train_test_split(\n    train_x, train_y, test_size=0.01, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:13:04.850452Z","iopub.execute_input":"2022-07-21T00:13:04.850869Z","iopub.status.idle":"2022-07-21T00:13:04.861044Z","shell.execute_reply.started":"2022-07-21T00:13:04.850827Z","shell.execute_reply":"2022-07-21T00:13:04.859841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(val_x, val_y, clf, cv=5):\n    print(f\"Val Score: {clf.score(val_x, val_y):.4f}\")\n    print(classification_report(val_y, clf.predict(val_x)))\n    cvs = cross_val_score(clf, val_x, val_y, cv=cv)\n    cvs_std = np.std(cvs)\n    cvs_mean = np.mean(cvs)\n    print(f\"Cross validation score: {cvs_mean:.4f}\")\n    print(f\"Cross validation range: {cvs_mean-(2*cvs_std):.4f}-{cvs_mean+(2*cvs_std):.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:13:04.862133Z","iopub.execute_input":"2022-07-21T00:13:04.863637Z","iopub.status.idle":"2022-07-21T00:13:04.871062Z","shell.execute_reply.started":"2022-07-21T00:13:04.863590Z","shell.execute_reply":"2022-07-21T00:13:04.869994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train model with train-test split\n\nShould probably move to cross validation eventually.....","metadata":{}},{"cell_type":"code","source":"clf = CatBoostClassifier(random_state=42, verbose=False)\n# clf.fit(train_x, train_y)\nds = Pool(train_x, label=train_y)\n\ngrid = {'learning_rate': [0.1, 0.01, 0.001, 0.0001],\n'depth': [5, 10, 20, 50, 100],\n'l2_leaf_reg': [1, 3, 5],\n'iterations': [100, 250, 500, 1000]}\nclf.grid_search(grid, ds)\nevaluate_model(val_x, val_y, clf)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:15:14.519310Z","iopub.execute_input":"2022-07-21T00:15:14.520298Z","iopub.status.idle":"2022-07-21T00:26:41.911587Z","shell.execute_reply.started":"2022-07-21T00:15:14.520254Z","shell.execute_reply":"2022-07-21T00:26:41.910374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_x = train_pp.transform(test_data)\ntest_data['Transported'] = [bool(x) for x in clf.predict(test_x).tolist()]\nplt.hist(clf.predict_proba(test_x)[:, 1])\nplt.title(\"Test Pred values\")\nplt.show()\nplt.hist(clf.predict(test_x))\nplt.title(\"Test class pred values\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T00:42:56.856461Z","iopub.execute_input":"2022-07-21T00:42:56.856877Z","iopub.status.idle":"2022-07-21T00:42:57.375439Z","shell.execute_reply.started":"2022-07-21T00:42:56.856841Z","shell.execute_reply":"2022-07-21T00:42:57.374376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make submission file","metadata":{}},{"cell_type":"code","source":"test_data[['PassengerId','Transported']].to_csv('sampleSubmission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:05.974394Z","iopub.execute_input":"2022-07-17T19:51:05.974808Z","iopub.status.idle":"2022-07-17T19:51:05.996946Z","shell.execute_reply.started":"2022-07-17T19:51:05.974774Z","shell.execute_reply":"2022-07-17T19:51:05.995606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train a neural net model\n","metadata":{}},{"cell_type":"code","source":"clf2 = MLPClassifier(random_state=42, verbose=False, max_iter=1000)\nclf2.fit(train_x, train_y)\nevaluate_model(val_x, val_y, clf2)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T01:50:27.837764Z","iopub.execute_input":"2022-07-20T01:50:27.838156Z","iopub.status.idle":"2022-07-20T01:52:15.739222Z","shell.execute_reply.started":"2022-07-20T01:50:27.838115Z","shell.execute_reply":"2022-07-20T01:52:15.737947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_x = train_pp.transform(test_data)\ntest_data['Transported'] = [bool(x) for x in clf2.predict(test_x).tolist()]\ntest_data[['PassengerId','Transported']].to_csv('sampleSubmission_nn.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T01:56:33.468318Z","iopub.execute_input":"2022-07-20T01:56:33.469405Z","iopub.status.idle":"2022-07-20T01:56:33.574175Z","shell.execute_reply.started":"2022-07-20T01:56:33.469359Z","shell.execute_reply":"2022-07-20T01:56:33.572871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensemble","metadata":{}},{"cell_type":"code","source":"eclf = VotingClassifier(estimators=[('catboost', clf), \n                                    ('nn', clf2), \n                                    ('rf', RandomForestClassifier()), \n                                    ('lr', LogisticRegression()),\n                                    ('hist', HistGradientBoostingClassifier()),\n                                    ('dt', DecisionTreeClassifier(max_depth=20)),\n                                    ('knn', KNeighborsClassifier(n_neighbors=7))\n                                   ],\n                         voting='soft', weights=[2, 1,1,1,1,1,1])\neclf.fit(train_x.toarray(), train_y)\nevaluate_model(val_x.toarray(), val_y, eclf)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T02:35:33.299139Z","iopub.execute_input":"2022-07-20T02:35:33.299705Z","iopub.status.idle":"2022-07-20T02:36:36.587246Z","shell.execute_reply.started":"2022-07-20T02:35:33.299653Z","shell.execute_reply":"2022-07-20T02:36:36.585819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_x = train_pp.transform(test_data)\ntest_data['Transported'] = [bool(x) for x in eclf.predict(test_x.toarray()).tolist()]\ntest_data[['PassengerId','Transported']].to_csv('sampleSubmission_kitchen_sink.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T01:56:45.298203Z","iopub.execute_input":"2022-07-20T01:56:45.298619Z","iopub.status.idle":"2022-07-20T01:56:45.419428Z","shell.execute_reply.started":"2022-07-20T01:56:45.298586Z","shell.execute_reply":"2022-07-20T01:56:45.417729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Play with feature importances\n\nWhat happens if we remove the N-least useful features?","metadata":{}},{"cell_type":"code","source":"importances = clf.get_feature_importance()\nplt.hist(importances)\nplt.title(\"Histogram of feature importances\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:09.195769Z","iopub.execute_input":"2022-07-17T19:51:09.196262Z","iopub.status.idle":"2022-07-17T19:51:09.384535Z","shell.execute_reply.started":"2022-07-17T19:51:09.196227Z","shell.execute_reply":"2022-07-17T19:51:09.383209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Why do manual work?\n\nWe don't have _too_ many features. Might as well just loop over removing the n-least important features for varying values of n and pick the best performing one","metadata":{}},{"cell_type":"code","source":"from copy import deepcopy\n\ndef find_best_importance_threshold(feature_removal_range):\n    \"\"\"Search over a range of number of possible removable features. Find and return the model that performs best.\"\"\"\n    \n    train_x_copy = deepcopy(train_x)\n    val_x_copy = deepcopy(val_x)\n    \n    try:\n        train_x_copy = np.asarray(train_x_copy.todense())\n        val_x_copy = np.asarray(val_x_copy.todense())\n    except:\n        pass\n    \n    outputs = []\n        \n    for num_feats in tqdm(feature_removal_range):\n        print(f'Testing with {num_feats} features')\n        # Get n least important features\n        removable_cols = np.argsort(importances)[:num_feats]\n        new_train_x = np.delete(train_x_copy, removable_cols, axis=1)\n        new_val_x = np.delete(val_x_copy, removable_cols, axis=1)\n        \n        # Train model\n        new_clf = CatBoostClassifier(random_state=42, verbose=False)\n        new_clf.fit(new_train_x, train_y)\n        \n        # Evaluate on validation set\n        val_score = new_clf.score(new_val_x, val_y)\n        cvs = np.mean(cross_val_score(new_clf, new_val_x, val_y, cv=5))\n        print(f\"Val score: {val_score}\")\n        print(classification_report(val_y, new_clf.predict(new_val_x)))\n        print(f\"Cross Val score: {cvs}\")\n        \n        # Store output\n        outputs.append({\n            \"n_feats\": num_feats,\n            \"cols\": removable_cols,\n            \"model\": new_clf,\n            \"score\": val_score,\n            \"cross_val_score\": cvs\n        })\n        \n    # Return best performing model\n    return list(sorted(outputs, key=lambda x: x[\"cross_val_score\"], reverse=True))[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:01:31.813566Z","iopub.execute_input":"2022-07-17T19:01:31.814125Z","iopub.status.idle":"2022-07-17T19:01:31.830687Z","shell.execute_reply.started":"2022-07-17T19:01:31.814085Z","shell.execute_reply":"2022-07-17T19:01:31.829508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model = find_best_importance_threshold(list(range(1,20)))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:01:32.729431Z","iopub.execute_input":"2022-07-17T19:01:32.732581Z","iopub.status.idle":"2022-07-17T19:09:57.293268Z","shell.execute_reply.started":"2022-07-17T19:01:32.732514Z","shell.execute_reply":"2022-07-17T19:09:57.292042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(best_model)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:09:57.295860Z","iopub.execute_input":"2022-07-17T19:09:57.296460Z","iopub.status.idle":"2022-07-17T19:09:57.304746Z","shell.execute_reply.started":"2022-07-17T19:09:57.296407Z","shell.execute_reply":"2022-07-17T19:09:57.303303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_x = train_pp.transform(test_data)\n\ntest_x_copy = deepcopy(test_x)\n\ntry:\n    test_x_copy = np.asarray(test_x_copy.todense())\nexcept:\n    pass\n\ntest_x_new = np.delete(test_x_copy, best_model['cols'], axis=1)\ntest_data['Transported'] = [bool(x) for x in best_model['model'].predict(test_x_new).tolist()]\nplt.hist(best_model['model'].predict_proba(test_x_new)[:, 1])\nplt.title(\"Test Pred values (with cleaned features)\")\nplt.show()\nplt.hist(best_model['model'].predict(test_x_new))\nplt.title(\"Test class pred values\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:09:57.306852Z","iopub.execute_input":"2022-07-17T19:09:57.307313Z","iopub.status.idle":"2022-07-17T19:09:58.023346Z","shell.execute_reply.started":"2022-07-17T19:09:57.307249Z","shell.execute_reply":"2022-07-17T19:09:58.022104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data[['PassengerId','Transported']].to_csv('sampleSubmission2.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:09:58.025603Z","iopub.execute_input":"2022-07-17T19:09:58.025967Z","iopub.status.idle":"2022-07-17T19:09:58.043486Z","shell.execute_reply.started":"2022-07-17T19:09:58.025934Z","shell.execute_reply":"2022-07-17T19:09:58.042250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}