{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd","metadata":{"id":"JIdTslyLui3C","execution":{"iopub.status.busy":"2022-08-05T11:34:55.192568Z","iopub.execute_input":"2022-08-05T11:34:55.193057Z","iopub.status.idle":"2022-08-05T11:34:55.199090Z","shell.execute_reply.started":"2022-08-05T11:34:55.193010Z","shell.execute_reply":"2022-08-05T11:34:55.198046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wget https://nkb-backend-otg-media-static.s3.ap-south-1.amazonaws.com/otg_prod/media/Tech_4.0/AI_ML/Datasets/kaggle_titanic/train.csv\n!wget https://nkb-backend-otg-media-static.s3.ap-south-1.amazonaws.com/otg_prod/media/Tech_4.0/AI_ML/Datasets/kaggle_titanic/test.csv","metadata":{"id":"1k9yeXlTutDg","outputId":"bbe21c90-7631-45a9-a2dc-6ecf152d2a7b","execution":{"iopub.status.busy":"2022-08-05T11:34:49.941107Z","iopub.execute_input":"2022-08-05T11:34:49.941533Z","iopub.status.idle":"2022-08-05T11:34:55.190266Z","shell.execute_reply.started":"2022-08-05T11:34:49.941481Z","shell.execute_reply":"2022-08-05T11:34:55.188823Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('train.csv')\ntest_X_df = pd.read_csv('test.csv')\n\ntrain_Y_df = train_df['Survived']\ntrain_X_df = train_df.drop(['Survived'], axis=1)","metadata":{"id":"urX-m2FCuyvj","execution":{"iopub.status.busy":"2022-08-05T11:35:03.062993Z","iopub.execute_input":"2022-08-05T11:35:03.063663Z","iopub.status.idle":"2022-08-05T11:35:03.082187Z","shell.execute_reply.started":"2022-08-05T11:35:03.063625Z","shell.execute_reply":"2022-08-05T11:35:03.081144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X_df","metadata":{"id":"qj8ZEGRXZpfK","outputId":"f7156d8b-3dd3-4ab9-b28d-4baa31cb9dbb","execution":{"iopub.status.busy":"2022-08-05T11:35:06.220950Z","iopub.execute_input":"2022-08-05T11:35:06.221376Z","iopub.status.idle":"2022-08-05T11:35:06.253180Z","shell.execute_reply.started":"2022-08-05T11:35:06.221339Z","shell.execute_reply":"2022-08-05T11:35:06.252268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['Cabin','PassengerId', 'Name', 'Ticket']\n#cols = ['PassengerId', 'Name', 'Ticket']\n\ntrain_X_df.drop(cols, axis=1, inplace=True)\npass_id = test_X_df['PassengerId']\ntest_X_df.drop(cols, axis=1, inplace=True)","metadata":{"id":"Z5IA5968u2PL","execution":{"iopub.status.busy":"2022-08-05T11:35:09.466624Z","iopub.execute_input":"2022-08-05T11:35:09.467022Z","iopub.status.idle":"2022-08-05T11:35:09.475829Z","shell.execute_reply.started":"2022-08-05T11:35:09.466990Z","shell.execute_reply":"2022-08-05T11:35:09.474756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X_df","metadata":{"id":"lZWkJw-5bxse","outputId":"771ecc06-957b-49bb-b750-c268c5a92ed8","execution":{"iopub.status.busy":"2022-08-05T11:35:11.455097Z","iopub.execute_input":"2022-08-05T11:35:11.455507Z","iopub.status.idle":"2022-08-05T11:35:11.476403Z","shell.execute_reply.started":"2022-08-05T11:35:11.455473Z","shell.execute_reply":"2022-08-05T11:35:11.474933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X_df.shape","metadata":{"id":"vAULxpKfPUt6","outputId":"44ddae96-daca-40a9-f6a3-6421128f3f21","execution":{"iopub.status.busy":"2022-08-05T11:35:14.541697Z","iopub.execute_input":"2022-08-05T11:35:14.542093Z","iopub.status.idle":"2022-08-05T11:35:14.548916Z","shell.execute_reply.started":"2022-08-05T11:35:14.542062Z","shell.execute_reply":"2022-08-05T11:35:14.547562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X_df.isna().sum()","metadata":{"id":"DKdaEvHkO-WL","outputId":"dbdae062-b6ef-4d35-f731-19134528c581","execution":{"iopub.status.busy":"2022-08-05T11:35:16.871692Z","iopub.execute_input":"2022-08-05T11:35:16.873008Z","iopub.status.idle":"2022-08-05T11:35:16.884200Z","shell.execute_reply.started":"2022-08-05T11:35:16.872954Z","shell.execute_reply":"2022-08-05T11:35:16.882512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Applying Pipelines on Titanic Data","metadata":{"id":"IQyHV0j8CvB8"}},{"cell_type":"markdown","source":"### For categorical data","metadata":{"id":"5a8v5e0Dm4lM"}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.pipeline import Pipeline\n\ncategorical_cols_1 = ['Embarked', 'Sex']\ncategorical_pipeline_1 = Pipeline(steps=[\n                                          ('imputer', SimpleImputer(strategy='most_frequent')),\n                                          ('onehot', OneHotEncoder(drop='first',sparse=False))\n                                   ])","metadata":{"id":"NAZl8eab-cOp","execution":{"iopub.status.busy":"2022-08-05T11:35:28.041240Z","iopub.execute_input":"2022-08-05T11:35:28.041663Z","iopub.status.idle":"2022-08-05T11:35:28.859356Z","shell.execute_reply.started":"2022-08-05T11:35:28.041625Z","shell.execute_reply":"2022-08-05T11:35:28.858245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade category_encoders\nfrom category_encoders import TargetEncoder","metadata":{"id":"IKrr4oISJ1xY","execution":{"iopub.status.busy":"2022-08-05T11:35:30.373721Z","iopub.execute_input":"2022-08-05T11:35:30.374102Z","iopub.status.idle":"2022-08-05T11:35:43.521850Z","shell.execute_reply.started":"2022-08-05T11:35:30.374072Z","shell.execute_reply":"2022-08-05T11:35:43.520529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical_cols_2 = ['Cabin']\n# categorical_pipeline_2 = Pipeline(steps=[\n#                                           ('imputer', SimpleImputer(strategy='constant',fill_value=0)),\n#                                           ('target', TargetEncoder(categorical_cols_2))\n#                                    ])\n","metadata":{"id":"C7Xd5QeAWuyq","execution":{"iopub.status.busy":"2022-08-05T11:35:43.524417Z","iopub.execute_input":"2022-08-05T11:35:43.524814Z","iopub.status.idle":"2022-08-05T11:35:43.529812Z","shell.execute_reply.started":"2022-08-05T11:35:43.524774Z","shell.execute_reply":"2022-08-05T11:35:43.528899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### For numerical data","metadata":{"id":"eMeoStyvm0Q2"}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\nnumerical_cols = ['Age','Fare']\nnumerical_pipeline = Pipeline(steps=[('imputer', SimpleImputer())])","metadata":{"id":"jpDa5B_E-FA2","execution":{"iopub.status.busy":"2022-08-05T11:35:56.041907Z","iopub.execute_input":"2022-08-05T11:35:56.042541Z","iopub.status.idle":"2022-08-05T11:35:56.048354Z","shell.execute_reply.started":"2022-08-05T11:35:56.042488Z","shell.execute_reply":"2022-08-05T11:35:56.047186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Using ColumnTransformer","metadata":{"id":"LDuPvXyJ0p8p"}},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\n\ncolumn_transformer = ColumnTransformer(\n                                        transformers=[\n                                            ('cat_1', categorical_pipeline_1, categorical_cols_1),\n                                           # ('cat_2',categorical_pipeline_2,categorical_cols_2),\n                                            ('num', numerical_pipeline, numerical_cols)\n                                            ], \n                                        remainder='passthrough'\n                                      )","metadata":{"id":"ntYCoxdU-6x7","execution":{"iopub.status.busy":"2022-08-05T11:36:03.958659Z","iopub.execute_input":"2022-08-05T11:36:03.959660Z","iopub.status.idle":"2022-08-05T11:36:03.971500Z","shell.execute_reply.started":"2022-08-05T11:36:03.959603Z","shell.execute_reply":"2022-08-05T11:36:03.970233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pipelines","metadata":{"id":"jlGVEaT73CLN"}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nfrom sklearn.feature_selection import SelectKBest\nfrom sklearn.linear_model import LogisticRegression\n\npipe = Pipeline(steps=[('preprocessor', column_transformer),\n                       ('scaler', RobustScaler()),\n                       ('selector', SelectKBest(k=5)),\n                       ('classifier', LogisticRegression(solver='liblinear', max_iter=2000))])\npipe.fit(train_X_df,train_Y_df)\n","metadata":{"id":"lXSy219pPxX3","outputId":"df0d80f0-19d5-4b86-f75e-1f9b7dd26c27","execution":{"iopub.status.busy":"2022-08-05T11:36:21.982500Z","iopub.execute_input":"2022-08-05T11:36:21.983096Z","iopub.status.idle":"2022-08-05T11:36:22.047320Z","shell.execute_reply.started":"2022-08-05T11:36:21.983056Z","shell.execute_reply":"2022-08-05T11:36:22.046410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_test_Y = pipe.predict(test_X_df)\npredicted_test_Y","metadata":{"id":"D6-4gFH6zD9t","execution":{"iopub.status.busy":"2022-08-05T11:36:27.000437Z","iopub.execute_input":"2022-08-05T11:36:27.001042Z","iopub.status.idle":"2022-08-05T11:36:27.018369Z","shell.execute_reply.started":"2022-08-05T11:36:27.001008Z","shell.execute_reply":"2022-08-05T11:36:27.017395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transformers as Hyperparameters","metadata":{"id":"u1izL4VuPxX1"}},{"cell_type":"markdown","source":"**Passing pipeline to RandomizedSearchCV**","metadata":{"id":"BX4kfSkjPxX4"}},{"cell_type":"markdown","source":"1. Individual steps of the pipeline can also be tuned as hyperparameters. \n    * Ex: `'scaler': [ StandardScaler(), MinMaxScaler(),..`\n    * In such case, `('scaler', RobustScaler())` in the initially defined pipeline is just a placeholder.\n2. Transformer steps can be ignored by setting them to `passthrough`.\n    * Ex: With `'scaler': [ StandardScaler(), MinMaxScaler(), RobustScaler(), 'passthrough']`, 'No scaling' will also be tried as an option along with others.\n3. Parameters of transformers can also be tuned as hyperparameters\n    * Ex: With `'selector__k': [3, 5, 7]`, <br>Parameter `k`(number of features to select) of `SelectKBest()` will be tuned.","metadata":{"id":"u-RcKmiEbyH2"}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler\nfrom sklearn.model_selection import RandomizedSearchCV\n\nparam_distributions = {\n    'scaler': [StandardScaler(), MinMaxScaler(), RobustScaler(), 'passthrough'],\n    'selector__k': [3,4, 5, 6, 7],\n    'classifier__penalty': ['l2', 'l1'],\n    'classifier__C':  [0.0001, 0.001, 0.01, 0.1, 0.5, 1.0]\n}\n\nrandom_search_cv = RandomizedSearchCV(pipe, param_distributions=param_distributions, n_iter=30, scoring='accuracy', refit=True, cv=5, random_state=0) \nrandom_search_cv.fit(train_X_df, train_Y_df)\nprint(random_search_cv.best_params_)","metadata":{"id":"YyxvM6cmPxX5","outputId":"56a6ff7d-22d0-4e20-99c7-2b60dd8ff976","execution":{"iopub.status.busy":"2022-08-05T11:36:30.644337Z","iopub.execute_input":"2022-08-05T11:36:30.645227Z","iopub.status.idle":"2022-08-05T11:36:34.117284Z","shell.execute_reply.started":"2022-08-05T11:36:30.645186Z","shell.execute_reply":"2022-08-05T11:36:34.116170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the above code,\n* Number of possible combinations of hyperparameters =\n\n    = 4 scaler options x 3 selector options x 2 penalty options x 6 'C' options\n\n    = 4 x 3 x 2 x 6 = **144** possible combinations\n* Out of 144 possible combinations, 30 randomly chosen combinations are tried (`n_iter = 30`)\n* Each of these 30 combinations is tried with 5-Fold Cross-Validation. <br>Hence, the total number of models that will be trained = 30 x 5 = **150**","metadata":{"id":"0hSUdywAlEAd"}},{"cell_type":"markdown","source":"## Estimators as Hyperparameters","metadata":{"id":"QDLeGu3Pi3Tn"}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\n\nparam_distributions = {\n    'scaler': [ StandardScaler(), MinMaxScaler(), RobustScaler(), 'passthrough'],\n    'selector__k': [3 ,5 ,6 ,7],\n    'classifier': [LogisticRegression(solver='liblinear', max_iter=2000), KNeighborsClassifier()]\n}\n\nrandom_search_cv = RandomizedSearchCV(pipe, param_distributions=param_distributions, n_iter=32, scoring='accuracy', refit=True, cv=5, random_state=0) \nrandom_search_cv.fit(train_X_df, train_Y_df)\nprint(random_search_cv.best_params_)","metadata":{"id":"1LeIBfXji1pg","outputId":"533c6153-2f3c-4338-9551-f39e0d278840","execution":{"iopub.status.busy":"2022-08-05T11:36:43.960100Z","iopub.execute_input":"2022-08-05T11:36:43.960511Z","iopub.status.idle":"2022-08-05T11:36:48.268817Z","shell.execute_reply.started":"2022-08-05T11:36:43.960478Z","shell.execute_reply":"2022-08-05T11:36:48.267923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_search_cv.cv_results_","metadata":{"id":"5B7hZXZdphy8","outputId":"b5e7735f-433b-4134-aa29-3f8d660f769a","execution":{"iopub.status.busy":"2022-08-05T11:36:48.270745Z","iopub.execute_input":"2022-08-05T11:36:48.271081Z","iopub.status.idle":"2022-08-05T11:36:48.310057Z","shell.execute_reply.started":"2022-08-05T11:36:48.271051Z","shell.execute_reply":"2022-08-05T11:36:48.308742Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the above code,\n* Number of possible combinations =\n\n    = 4 scaler options x 3 selector options x 2 classifier options\n\n    = 4 x 3 x 2 = **24** possible combinations\n* Out of 24 possible combinations, 20 randomly chosen combinations are tried (`n_iter` = 20)\n* Each of these 20 combinations is tried with 5-Fold Cross-Validation. <br>Hence, the total number of models that will be trained = 20 x 5 = **100**","metadata":{"id":"OX-ErAJqqKAy"}},{"cell_type":"markdown","source":"## List of Dicts as Parameter Distributions\n\nIf we want to tune the hyperparameters of individual estimators separately, a list of dictionaries can be passed as an argument to  `param_distributions` parameter. This can be used in both `RandomizedSearchCV` or `GridSearchCV` appropriately as shown below.","metadata":{"id":"lkbTmMxgTYGc"}},{"cell_type":"markdown","source":"**Pipeline**","metadata":{"id":"YNO9LNA-TYGe"}},{"cell_type":"code","source":"pipe = Pipeline(steps=[('preprocessor', column_transformer),\n                       ('scaler', RobustScaler()),\n                       ('selector', SelectKBest(k=5)),\n                       ('classifier', LogisticRegression( max_iter=2000))])","metadata":{"id":"VH3VNUBJTYGe","execution":{"iopub.status.busy":"2022-08-05T11:36:48.312295Z","iopub.execute_input":"2022-08-05T11:36:48.313014Z","iopub.status.idle":"2022-08-05T11:36:48.319109Z","shell.execute_reply.started":"2022-08-05T11:36:48.312978Z","shell.execute_reply":"2022-08-05T11:36:48.318146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe = Pipeline(steps=[('preprocessor', column_transformer),\n                       ('scaler', RobustScaler()),\n                       ('selector', SelectKBest(k=5)),\n                       ('classifier', KNeighborsClassifier(n_neighbors=11,weights='distance'))])\npipe.fit(train_X_df,train_Y_df)\n","metadata":{"id":"rC0-6nO7G7eo","outputId":"c63ae149-23bb-407a-d515-2d0341f40ad8","execution":{"iopub.status.busy":"2022-08-05T11:36:49.642106Z","iopub.execute_input":"2022-08-05T11:36:49.642975Z","iopub.status.idle":"2022-08-05T11:36:49.692899Z","shell.execute_reply.started":"2022-08-05T11:36:49.642927Z","shell.execute_reply":"2022-08-05T11:36:49.691580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_test_Y = pipe.predict(test_X_df)\n#predicted_test_Y","metadata":{"id":"l-kL_Pa2g7N_","execution":{"iopub.status.busy":"2022-08-05T11:36:52.140863Z","iopub.execute_input":"2022-08-05T11:36:52.141281Z","iopub.status.idle":"2022-08-05T11:36:52.159043Z","shell.execute_reply.started":"2022-08-05T11:36:52.141246Z","shell.execute_reply":"2022-08-05T11:36:52.157697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Passing pipeline to RandomizedSearchCV**","metadata":{"id":"EF3plMaSTYGe"}},{"cell_type":"code","source":"param_distributions = [\n              {\n                'scaler': [StandardScaler(), MinMaxScaler(), RobustScaler(), 'passthrough'],\n                'classifier': [LogisticRegression(solver='liblinear', max_iter=2000)],\n                'classifier__penalty': ['l2', 'l1'],\n                 'selector__k': [3, 4, 5, 6,7,8],\n                 'classifier__C':  [0.0001, 0.001,0.005, 0.01,0.05, 0.1, 0.5, 1.0]\n               \n              },\n              {\n                'scaler': [StandardScaler(), MinMaxScaler(), RobustScaler(), 'passthrough'],\n                 'selector__k': [3, 4, 5, 6,7,8],\n                 'classifier': [KNeighborsClassifier()],\n                'classifier__n_neighbors':  range(1,20),\n                'classifier__p': [ 2, 3, 4]\n              }\n            ]\n\nrandom_search_cv = RandomizedSearchCV(pipe, param_distributions=param_distributions, n_iter=324, scoring='accuracy', refit=True, cv=5, random_state=0) \nrandom_search_cv.fit(train_X_df, train_Y_df)\nprint(random_search_cv.best_params_)","metadata":{"id":"-q7DHzlbTYGf","outputId":"00796c86-ab87-425e-e1a7-bc492770a446","execution":{"iopub.status.busy":"2022-08-05T11:36:53.838356Z","iopub.execute_input":"2022-08-05T11:36:53.838814Z","iopub.status.idle":"2022-08-05T11:37:42.346960Z","shell.execute_reply.started":"2022-08-05T11:36:53.838778Z","shell.execute_reply":"2022-08-05T11:37:42.345602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_search_cv.cv_results_['params']","metadata":{"id":"pbg264-nvOlM","execution":{"iopub.status.busy":"2022-08-05T11:37:52.630767Z","iopub.execute_input":"2022-08-05T11:37:52.631168Z","iopub.status.idle":"2022-08-05T11:37:52.797078Z","shell.execute_reply.started":"2022-08-05T11:37:52.631136Z","shell.execute_reply":"2022-08-05T11:37:52.796134Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Prediction on Test Data**","metadata":{"id":"M0w-GT4STYGf"}},{"cell_type":"code","source":"best_model = random_search_cv.best_estimator_\npredicted_test_Y = best_model.predict(test_X_df)\n#predicted_test_Y","metadata":{"id":"5MGUGn2tTYGf","execution":{"iopub.status.busy":"2022-08-05T11:38:10.904167Z","iopub.execute_input":"2022-08-05T11:38:10.904675Z","iopub.status.idle":"2022-08-05T11:38:10.935028Z","shell.execute_reply.started":"2022-08-05T11:38:10.904628Z","shell.execute_reply":"2022-08-05T11:38:10.933647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_test_Y_df = pd.DataFrame(predicted_test_Y)\npass_id_df=pd.DataFrame(pass_id)\nresult = pass_id_df.assign( Survived = predicted_test_Y_df)\n#pd.DataFrame(result).to_csv('result.csv', header=True, index=False)\npd.DataFrame(result).to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"id":"gZjgiFBBvGmw","execution":{"iopub.status.busy":"2022-08-05T11:50:37.170030Z","iopub.execute_input":"2022-08-05T11:50:37.170429Z","iopub.status.idle":"2022-08-05T11:50:37.181184Z","shell.execute_reply.started":"2022-08-05T11:50:37.170396Z","shell.execute_reply":"2022-08-05T11:50:37.180015Z"},"trusted":true},"execution_count":null,"outputs":[]}]}