{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import classification_report\nfrom sklearn.model_selection import RandomizedSearchCV\nimport seaborn as sns\npd.set_option('display.max_columns', 500)","metadata":{"id":"b5jh6UzxWsLw","execution":{"iopub.status.busy":"2022-07-15T04:35:45.159300Z","iopub.execute_input":"2022-07-15T04:35:45.159881Z","iopub.status.idle":"2022-07-15T04:35:48.017007Z","shell.execute_reply.started":"2022-07-15T04:35:45.159795Z","shell.execute_reply":"2022-07-15T04:35:48.015591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://www.svds.com/learning-imbalanced-classes/\n\nhttps://machinelearningmastery.com/tour-of-evaluation-metrics-for-imbalanced-classification/\n\nhttps://towardsdatascience.com/one-common-misconception-about-random-forest-and-overfitting-47cae2e2c23b","metadata":{"id":"yXwY8eE2TKRu"}},{"cell_type":"markdown","source":"https://neptune.ai/blog/evaluation-metrics-binary-classification\n\nhttps://medium.com/usf-msds/choosing-the-right-metric-for-evaluating-machine-learning-models-part-2-86d5649a5428\n\nhttps://towardsdatascience.com/silhouette-method-better-than-elbow-method-to-find-optimal-clusters-378d62ff6891","metadata":{"id":"nJJANUZlVUA0"}},{"cell_type":"code","source":"# To Import Data \nData=pd.read_csv('../input/customer-churn-prediction-2020/train.csv')\ndf_Test = pd.read_csv('../input/customer-churn-prediction-2020/test.csv')\nTest_Data_identifier = df_Test[['id']]\ndf_Test.drop( columns='id', inplace=True )","metadata":{"id":"47y_40JEWjro","outputId":"4f3108cc-57fa-4410-f773-9325dd6ab307","execution":{"iopub.status.busy":"2022-07-15T04:35:48.019208Z","iopub.execute_input":"2022-07-15T04:35:48.019552Z","iopub.status.idle":"2022-07-15T04:35:48.070163Z","shell.execute_reply.started":"2022-07-15T04:35:48.019529Z","shell.execute_reply":"2022-07-15T04:35:48.068985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Features**:\n***\n\nstate, string --> 2-letter code of the US state of customer residence <br>\naccount_length, numerical --> Number of months the customer has been with the current telco provider <br>\narea_code --> string=\"area_code_AAA\" where AAA = 3 digit area code.<br>\ninternational_plan, (yes/no) --> The customer has international plan.<br>\nvoice_mail_plan, (yes/no) --> The customer has voice mail plan.<br>\nnumber_vmail_messages, numerical --> Number of voice-mail messages.<br>\ntotal_day_minutes, numerical.--> Total minutes of day calls.<br>\ntotal_day_calls, numerical.--> Total number of day calls.<br>\ntotal_day_charge, numerical. -->Total charge of day calls.<br>\ntotal_eve_minutes, numerical.--> Total minutes of evening calls.<br>\ntotal_eve_calls, numerical. -->Total number of evening calls.<br>\ntotal_eve_charge, numerical. -->Total charge of evening calls.<br>\ntotal_night_minutes, numerical. -->Total minutes of night calls.<br>\ntotal_night_calls, numerical.--> Total number of night calls.<br>\ntotal_night_charge, numerical. --> Total charge of night calls.<br>\ntotal_intl_minutes, numerical. --> Total minutes of international calls.<br>\ntotal_intl_calls, numerical.--> Total number of international calls.<br>\ntotal_intl_charge, numerical.--> Total charge of international calls. <br>\nnumber_customer_service_calls, numerical. --> Number of calls to customer service.<br>\nchurn, (yes/no).--> Customer churn - **target** variable.<br>\n\n\n","metadata":{"id":"jwYsfdFHYY4o"}},{"cell_type":"code","source":"target_churn = Data[['churn']]\ndf = Data.drop(columns='churn',axis=1).copy()\nData.head()","metadata":{"id":"VxvGz5eXXfzm","outputId":"b1c6d2a7-6c9c-4e64-884c-fd65601112e1","execution":{"iopub.status.busy":"2022-07-15T04:35:48.071615Z","iopub.execute_input":"2022-07-15T04:35:48.072359Z","iopub.status.idle":"2022-07-15T04:35:48.113921Z","shell.execute_reply.started":"2022-07-15T04:35:48.072319Z","shell.execute_reply":"2022-07-15T04:35:48.112747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_features_name = ['state','area_code','international_plan','voice_mail_plan']\nnumerical_features_name = list( set(df.columns.values) - set(categorical_features_name) )","metadata":{"id":"nPOrrew2J1UF","execution":{"iopub.status.busy":"2022-07-15T04:35:48.117310Z","iopub.execute_input":"2022-07-15T04:35:48.118026Z","iopub.status.idle":"2022-07-15T04:35:48.137626Z","shell.execute_reply.started":"2022-07-15T04:35:48.117955Z","shell.execute_reply":"2022-07-15T04:35:48.136719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features_name","metadata":{"id":"p4piSrGWKiNe","outputId":"2bcd5f3f-420f-4a3a-c0b8-d046292b2dd8","execution":{"iopub.status.busy":"2022-07-15T04:35:48.138724Z","iopub.execute_input":"2022-07-15T04:35:48.139102Z","iopub.status.idle":"2022-07-15T04:35:48.173224Z","shell.execute_reply.started":"2022-07-15T04:35:48.139059Z","shell.execute_reply":"2022-07-15T04:35:48.172287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['area_code'].unique() # Categorical Variable","metadata":{"id":"rSKnpQVYXgYO","outputId":"5a98bd5e-f40d-4fb0-8a02-96ce1cd12da2","execution":{"iopub.status.busy":"2022-07-15T04:35:48.174284Z","iopub.execute_input":"2022-07-15T04:35:48.174696Z","iopub.status.idle":"2022-07-15T04:35:48.188763Z","shell.execute_reply.started":"2022-07-15T04:35:48.174661Z","shell.execute_reply":"2022-07-15T04:35:48.187944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['state'].unique() # Categorical Variable","metadata":{"id":"J-USEKDU4DPW","outputId":"ef3b620e-b9a3-4b2a-f48d-8f584a996bbf","execution":{"iopub.status.busy":"2022-07-15T04:35:48.190236Z","iopub.execute_input":"2022-07-15T04:35:48.190551Z","iopub.status.idle":"2022-07-15T04:35:48.203309Z","shell.execute_reply.started":"2022-07-15T04:35:48.190515Z","shell.execute_reply":"2022-07-15T04:35:48.202415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"id":"jcJAEDXQ40G-","outputId":"a50f0d3e-937e-431e-d338-a312e8995ff4","execution":{"iopub.status.busy":"2022-07-15T04:35:48.204759Z","iopub.execute_input":"2022-07-15T04:35:48.205072Z","iopub.status.idle":"2022-07-15T04:35:48.218108Z","shell.execute_reply.started":"2022-07-15T04:35:48.205042Z","shell.execute_reply":"2022-07-15T04:35:48.217282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization","metadata":{"id":"mjFlUwtfJqHd"}},{"cell_type":"code","source":"import seaborn as sns","metadata":{"id":"Dm23sQZb5aIs","execution":{"iopub.status.busy":"2022-07-15T04:35:48.219429Z","iopub.execute_input":"2022-07-15T04:35:48.219967Z","iopub.status.idle":"2022-07-15T04:35:48.225700Z","shell.execute_reply.started":"2022-07-15T04:35:48.219934Z","shell.execute_reply":"2022-07-15T04:35:48.224826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yes_counts = target_churn.value_counts()['yes']\nno_counts = target_churn.value_counts()['no']\nplt.pie( [yes_counts, no_counts], labels=['yes','no'] )\nprint( target_churn.value_counts() )\nprint('-----------------------------')\nprint('Percentage of Churn Yes', yes_counts*100/target_churn.shape[0] )","metadata":{"id":"5gbYhD0_MUXu","outputId":"ec7ecad5-75b7-4a57-e09f-bdf00ef9a3c4","execution":{"iopub.status.busy":"2022-07-15T04:35:48.229185Z","iopub.execute_input":"2022-07-15T04:35:48.231137Z","iopub.status.idle":"2022-07-15T04:35:48.375557Z","shell.execute_reply.started":"2022-07-15T04:35:48.231103Z","shell.execute_reply":"2022-07-15T04:35:48.373875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Target Variable has Imbalanced Distribution","metadata":{"id":"W8ZkdXPv1ciB"}},{"cell_type":"code","source":"fig = go.Figure()\nfig.add_trace( go.Histogram( x = df['international_plan'] , name='international_plan' ) )\nfig.add_trace( go.Histogram( x = df['voice_mail_plan'] , name='voice_mail_plan' ) )\nfig.show()","metadata":{"id":"KboU6muu8SaW","outputId":"01a4478c-d941-4e50-ea3c-7f87295b9980","execution":{"iopub.status.busy":"2022-07-15T04:35:48.381471Z","iopub.execute_input":"2022-07-15T04:35:48.382438Z","iopub.status.idle":"2022-07-15T04:35:48.537051Z","shell.execute_reply.started":"2022-07-15T04:35:48.382404Z","shell.execute_reply":"2022-07-15T04:35:48.535940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_temp = df[['state','international_plan','voice_mail_plan']]\n\nfig = go.Figure()\nfig.add_trace( go.Histogram( x = df_temp[ df_temp['international_plan']=='no' ]['state'] , name='No International Plan' ) )\nfig.add_trace( go.Histogram( x = df_temp[ df_temp['international_plan']=='yes' ]['state'] , name='Yes International Plan' ) )\nfig.add_trace( go.Histogram( x = df_temp[ df_temp['voice_mail_plan']=='no' ]['state'] , name='No Voice Mail Plan' ) )\nfig.add_trace( go.Histogram( x = df_temp[ df_temp['voice_mail_plan']=='yes' ]['state'] , name='Yes Voice Mail Plan' ) )","metadata":{"id":"989Upyw_DFGX","outputId":"35927f93-bdfb-4a74-f995-4495cbe71a79","execution":{"iopub.status.busy":"2022-07-15T04:35:48.538165Z","iopub.execute_input":"2022-07-15T04:35:48.538521Z","iopub.status.idle":"2022-07-15T04:35:48.589842Z","shell.execute_reply.started":"2022-07-15T04:35:48.538484Z","shell.execute_reply":"2022-07-15T04:35:48.589290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram( df, x='state', color='international_plan' )\nfig.show()","metadata":{"id":"ZL8YTXKDGda-","outputId":"ad24f0ac-63f4-449b-ea1b-66a3b3b264a2","execution":{"iopub.status.busy":"2022-07-15T04:35:48.590790Z","iopub.execute_input":"2022-07-15T04:35:48.591696Z","iopub.status.idle":"2022-07-15T04:35:49.788565Z","shell.execute_reply.started":"2022-07-15T04:35:48.591670Z","shell.execute_reply":"2022-07-15T04:35:49.787174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram( df, x='state', color='voice_mail_plan' )\nfig.show()","metadata":{"id":"CF8V3LzkGyNN","outputId":"bbb2913b-50fc-4c53-c7de-2cfdeb403bf4","execution":{"iopub.status.busy":"2022-07-15T04:35:49.790528Z","iopub.execute_input":"2022-07-15T04:35:49.790867Z","iopub.status.idle":"2022-07-15T04:35:49.893047Z","shell.execute_reply.started":"2022-07-15T04:35:49.790838Z","shell.execute_reply":"2022-07-15T04:35:49.891938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from plotly.subplots import make_subplots\n\ndf_temp = df[['area_code','international_plan','voice_mail_plan']]\nfig = make_subplots(rows=1,cols=2)\n\nfig.add_trace( go.Histogram( x = df_temp[ df_temp['international_plan']=='no' ]['area_code'] , name='No International Plan' ), row=1, col=1 )\nfig.add_trace( go.Histogram( x = df_temp[ df_temp['international_plan']=='yes' ]['area_code'] , name='Yes International Plan' ),row=1, col=1 )\nfig.add_trace( go.Histogram( x = df_temp[ df_temp['voice_mail_plan']=='no' ]['area_code'] , name='No Voice Mail Plan' ), row=1, col=2 )\nfig.add_trace( go.Histogram( x = df_temp[ df_temp['voice_mail_plan']=='yes' ]['area_code'] , name='Yes Voice Mail Plan' ),row=1, col=2 )","metadata":{"id":"tEGtykRAG_6U","outputId":"2a6bc3a1-bb39-4c04-c0d2-f8a11ecd1214","execution":{"iopub.status.busy":"2022-07-15T04:35:49.894332Z","iopub.execute_input":"2022-07-15T04:35:49.894660Z","iopub.status.idle":"2022-07-15T04:35:50.008806Z","shell.execute_reply.started":"2022-07-15T04:35:49.894629Z","shell.execute_reply":"2022-07-15T04:35:50.007744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encoding_plan(df):\n  df['international_plan'].replace( to_replace='no', value=0,inplace=True )\n  df['international_plan'].replace( to_replace='yes', value=1,inplace=True )\n  df['voice_mail_plan'].replace( to_replace='no', value=0, inplace=True )\n  df['voice_mail_plan'].replace( to_replace='yes', value=1, inplace=True )\n  return df\n\ndf = encoding_plan(df)\ndf_Test = encoding_plan(df_Test)\n\n# Churn ---> 1 | Chrun ---> 0\ndef encoding_churn(df):\n  df['churn'].replace( to_replace='yes', value=1, inplace=True )\n  df['churn'].replace( to_replace='no', value=0, inplace=True )\n  return df\ntarget_churn = encoding_churn(target_churn)","metadata":{"id":"ufntJusy8SgO","outputId":"05e4e9eb-0e4e-4ee4-8ab4-2f0666b5d8b3","execution":{"iopub.status.busy":"2022-07-15T04:35:50.009890Z","iopub.execute_input":"2022-07-15T04:35:50.010561Z","iopub.status.idle":"2022-07-15T04:35:50.034068Z","shell.execute_reply.started":"2022-07-15T04:35:50.010532Z","shell.execute_reply":"2022-07-15T04:35:50.033223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Unbalanced class, but one class if more important that the other. For e.g. in Fraud detection, it is more important to correctly label an instance as fraudulent, as opposed to labeling the non-fraudulent one. In this case, I would pick the classifier that has a good F1 score only on the important class. Recall that the F1-score is available per class.\n<br>\n\n* ## In this Churn \"Yes\" is more important to us so we will choose the model that provide high F1_Score for \"Yes\" ( i.e 1 as output )","metadata":{"id":"nCeXrxdlpYnB"}},{"cell_type":"code","source":"# from sklearn.preprocessing import OneHotEncoder\n# def One_Hot_Encoder (df,feature):\n#   ohe = OneHotEncoder(sparse=False,drop='first')\n#   ohe.fit( df[[feature]] )\n#   temp = ohe.transform( df[[feature]] )\n#   temp = pd.DataFrame( temp )\n#   df.drop(columns=feature,axis=1,inplace=True)\n#   df = pd.concat( [df,temp],axis=1 )\n#   return df\n\n# df = One_Hot_Encoder(df,feature='area_code')\n\n# df_Test = One_Hot_Encoder( df_Test, feature='area_code' )","metadata":{"id":"AIciqj4u-OVC","execution":{"iopub.status.busy":"2022-07-15T04:35:50.036303Z","iopub.execute_input":"2022-07-15T04:35:50.037668Z","iopub.status.idle":"2022-07-15T04:35:50.047745Z","shell.execute_reply.started":"2022-07-15T04:35:50.037626Z","shell.execute_reply":"2022-07-15T04:35:50.046473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\ndef One_Hot_Encoder (df,feature):\n  ohe = OneHotEncoder(sparse=False,drop='first')\n  ohe.fit( df[[feature]] )\n  temp = ohe.transform( df[[feature]] )\n  temp = pd.DataFrame( temp )\n  df.drop(columns=feature,axis=1,inplace=True)\n  df = pd.concat( [df,temp],axis=1 )\n  return df, ohe\n\ndf, ohe = One_Hot_Encoder(df,feature='area_code')\n\ndf_Test, ohe_df_Test = One_Hot_Encoder( df_Test, feature='area_code' )","metadata":{"id":"1mV-QuvJuCnG","execution":{"iopub.status.busy":"2022-07-15T04:35:50.050054Z","iopub.execute_input":"2022-07-15T04:35:50.050293Z","iopub.status.idle":"2022-07-15T04:35:50.075410Z","shell.execute_reply.started":"2022-07-15T04:35:50.050269Z","shell.execute_reply":"2022-07-15T04:35:50.073843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns.values","metadata":{"id":"Y6JW4aNUHFVX","outputId":"90381b0f-1c7f-4116-8580-5553e41b9e10","execution":{"iopub.status.busy":"2022-07-15T04:35:50.077427Z","iopub.execute_input":"2022-07-15T04:35:50.079238Z","iopub.status.idle":"2022-07-15T04:35:50.086993Z","shell.execute_reply.started":"2022-07-15T04:35:50.079189Z","shell.execute_reply":"2022-07-15T04:35:50.085724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"id":"EyGCJdgJd71k","outputId":"c1ba24cc-14d6-4d50-c72b-86ce4d76d97a","execution":{"iopub.status.busy":"2022-07-15T04:35:50.088595Z","iopub.execute_input":"2022-07-15T04:35:50.089285Z","iopub.status.idle":"2022-07-15T04:35:50.170216Z","shell.execute_reply.started":"2022-07-15T04:35:50.089249Z","shell.execute_reply":"2022-07-15T04:35:50.169210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = df.drop( columns = ['total_day_minutes','total_eve_minutes','total_night_minutes','total_intl_minutes'] )\n# numerical_features_name = list( set(numerical_features_name) - set( ['total_day_minutes','total_eve_minutes','total_night_minutes','total_intl_minutes'] ) )\n\n# Dropping Highly Coorelated feature is decreasing models performance","metadata":{"id":"82lqfD72YE5h","execution":{"iopub.status.busy":"2022-07-15T04:35:50.172562Z","iopub.execute_input":"2022-07-15T04:35:50.173281Z","iopub.status.idle":"2022-07-15T04:35:50.177448Z","shell.execute_reply.started":"2022-07-15T04:35:50.173251Z","shell.execute_reply":"2022-07-15T04:35:50.176494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train Test Split\nfrom sklearn.model_selection import train_test_split\nX = df.drop(columns='state')\ndf_Test = df_Test.drop( columns='state' )\ny = target_churn\nX_train, X_test, y_train ,y_test = train_test_split( X, y, test_size=0.2,  random_state=10, stratify=y ) \nX_train.shape, y_train.shape, X_test.shape, y_test.shape","metadata":{"id":"GMx380Kg655-","outputId":"201821ca-81a9-412a-8461-be6504c65654","execution":{"iopub.status.busy":"2022-07-15T04:35:50.178810Z","iopub.execute_input":"2022-07-15T04:35:50.179452Z","iopub.status.idle":"2022-07-15T04:35:50.224067Z","shell.execute_reply.started":"2022-07-15T04:35:50.179422Z","shell.execute_reply":"2022-07-15T04:35:50.222924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features_name","metadata":{"id":"NJC9MIljIuVe","outputId":"f8062c9d-4b69-4983-ce4e-79e8406cde31","execution":{"iopub.status.busy":"2022-07-15T04:35:50.225648Z","iopub.execute_input":"2022-07-15T04:35:50.226002Z","iopub.status.idle":"2022-07-15T04:35:50.233716Z","shell.execute_reply.started":"2022-07-15T04:35:50.225966Z","shell.execute_reply":"2022-07-15T04:35:50.232514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i=0\nX_num_train = X_train[numerical_features_name]\nfig = make_subplots( rows=1, cols = X_num_train.shape[1],subplot_titles = X_num_train.columns.values )\nfor feature in X_num_train.columns.values:\n  fig.add_trace( go.Histogram( x=X_num_train[feature]), row=1,col=i+1 )\n  i = i+1\n\nfig.update_layout( width=10000 )\nfig.show()","metadata":{"id":"zH0Py-6ILQdD","outputId":"e11687e1-e3ea-492c-d068-744d20b498d6","execution":{"iopub.status.busy":"2022-07-15T04:35:50.234975Z","iopub.execute_input":"2022-07-15T04:35:50.235635Z","iopub.status.idle":"2022-07-15T04:35:50.546336Z","shell.execute_reply.started":"2022-07-15T04:35:50.235602Z","shell.execute_reply":"2022-07-15T04:35:50.545216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i=0\nX_num_train = X_train[numerical_features_name]\nfig = make_subplots( rows=1, cols = X_num_train.shape[1],subplot_titles = X_num_train.columns.values )\nfor feature in X_num_train.columns.values:\n  fig.add_trace( go.Box( y=X_num_train[feature]), row=1,col=i+1 )\n  i = i+1\n\nfig.update_layout( width=5500, height=700 )\nfig.show()","metadata":{"id":"zF7hIW4gRC4s","outputId":"02e11033-ea3c-4975-dc38-d4e2db7f556c","execution":{"iopub.status.busy":"2022-07-15T04:35:50.547663Z","iopub.execute_input":"2022-07-15T04:35:50.547988Z","iopub.status.idle":"2022-07-15T04:35:50.720963Z","shell.execute_reply.started":"2022-07-15T04:35:50.547957Z","shell.execute_reply":"2022-07-15T04:35:50.720323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Almost all features have outliers, so for scaling we will choose RobustScaler","metadata":{"id":"yuIyxNBfTgSR"}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nRS = RobustScaler()\nRS.fit(X_train[numerical_features_name].values)\nX_train[numerical_features_name] = RS.transform(X_train[numerical_features_name])\nX_test[numerical_features_name] = RS.transform( X_test[numerical_features_name] )\ndf_Test[numerical_features_name] = RS.transform( df_Test[numerical_features_name] )","metadata":{"id":"DalHNXePGrn_","outputId":"c0761e83-fd74-4039-88f0-3088aae074c0","execution":{"iopub.status.busy":"2022-07-15T04:35:50.721822Z","iopub.execute_input":"2022-07-15T04:35:50.722664Z","iopub.status.idle":"2022-07-15T04:35:50.747339Z","shell.execute_reply.started":"2022-07-15T04:35:50.722632Z","shell.execute_reply":"2022-07-15T04:35:50.746373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train[numerical_features_name].values","metadata":{"id":"w8y5kITcnHTo","outputId":"212c33a6-1977-4b2d-fa5b-233f8da3343b","execution":{"iopub.status.busy":"2022-07-15T04:35:50.748799Z","iopub.execute_input":"2022-07-15T04:35:50.749455Z","iopub.status.idle":"2022-07-15T04:35:50.758175Z","shell.execute_reply.started":"2022-07-15T04:35:50.749414Z","shell.execute_reply":"2022-07-15T04:35:50.757429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i=0\n# X_num_train = X_train[numerical_features_name]\n# fig = make_subplots( rows=1, cols = X_num_train.shape[1],subplot_titles = X_num_train.columns.values )\n# for feature in X_num_train.columns.values:\n#   fig.add_trace( go.Histogram( x=X_num_train[feature]), row=1,col=i+1 )\n#   i = i+1\n\n# fig.update_layout( width=10000 )\n# fig.show()","metadata":{"id":"fvHTtMaoTp6E","execution":{"iopub.status.busy":"2022-07-15T04:35:50.759547Z","iopub.execute_input":"2022-07-15T04:35:50.760235Z","iopub.status.idle":"2022-07-15T04:35:50.769834Z","shell.execute_reply.started":"2022-07-15T04:35:50.760203Z","shell.execute_reply":"2022-07-15T04:35:50.769062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score, roc_curve\nfrom sklearn.metrics import f1_score\nfrom xgboost import XGBClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.tree import ExtraTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import RidgeClassifier\nfrom sklearn import svm, tree, linear_model, neighbors, naive_bayes, ensemble\nfrom sklearn import naive_bayes\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC","metadata":{"id":"sTSsTPzk94MM","execution":{"iopub.status.busy":"2022-07-15T04:35:50.770826Z","iopub.execute_input":"2022-07-15T04:35:50.771342Z","iopub.status.idle":"2022-07-15T04:35:51.110713Z","shell.execute_reply.started":"2022-07-15T04:35:50.771319Z","shell.execute_reply":"2022-07-15T04:35:51.109700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Evaluation(model,X_train,X_test,y_train,y_test,grid=True):\n  plt.figure(  figsize=(12,6) )\n  print( \"-----------------------------------------------------------------------------------------------------------\")\n  print( model )\n  print( \" For Train Set :  \")\n  y_pred = model.predict(X_train)\n  if grid==True:\n    print(\"Param for GS\", model.best_params_)\n    print(\"CV score for GS\", model.best_score_)\n\n  print(\"F1_Score = \", f1_score(y_train, y_pred, average='macro'))\n  print(\"AUC ROC Score for GS: \", roc_auc_score(y_train, y_pred))\n  print( classification_report( y_train, y_pred ) )\n\n  print( \" For Test Set :  \")\n  y_pred = model.predict(X_test)\n  print(\"Test Set F1_Score = \", f1_score(y_test, y_pred, average='macro'))\n  auc_score = roc_auc_score(y_test, y_pred )\n  print(\"Test AUC ROC Score for GS: \", auc_score )\n  print( classification_report( y_test, y_pred ) )\n  print('------------------------------------------------------------------------------------------------------------')\n  print(\"\\n\")\n\n  fpr, tpr, thresh = roc_curve( y_test, y_pred )\n  # plt.plot( fpr, tpr, label=model + str( auc_score ) ) \n  return","metadata":{"id":"zbCtpeH7CfZM","execution":{"iopub.status.busy":"2022-07-15T04:35:51.112174Z","iopub.execute_input":"2022-07-15T04:35:51.112456Z","iopub.status.idle":"2022-07-15T04:35:51.122425Z","shell.execute_reply.started":"2022-07-15T04:35:51.112428Z","shell.execute_reply":"2022-07-15T04:35:51.121789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_models_with_default_paramters(X_train,X_test,y_train,y_test):\n  models_default = [ RandomForestClassifier(), DecisionTreeClassifier(), XGBClassifier() , ExtraTreeClassifier(), KNeighborsClassifier(), RidgeClassifier(), SVC(), \n                     LogisticRegression(),AdaBoostClassifier(), SGDClassifier() ]\n\n  for model in models_default:\n    print(model)\n    model.fit(X_train, y_train)\n\n    Evaluation(model,X_train,X_test,y_train,y_test,False)\n\napply_models_with_default_paramters(X_train,X_test,y_train,y_test)","metadata":{"id":"cIc87-9K6XzL","outputId":"908a1c67-8202-4be1-bcef-0e9f593bd562","execution":{"iopub.status.busy":"2022-07-15T04:35:51.123300Z","iopub.execute_input":"2022-07-15T04:35:51.124096Z","iopub.status.idle":"2022-07-15T04:35:54.558679Z","shell.execute_reply.started":"2022-07-15T04:35:51.124070Z","shell.execute_reply":"2022-07-15T04:35:54.557850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* With default parameters, best performance is given by RandomForestClassifier\n* All the tree based models are overfitting ","metadata":{"id":"Lw01IvLKKcii"}},{"cell_type":"code","source":"def applying_hypertuning_models( X_train,X_test,y_train,y_test ):\n  # Decision Tree\n  param_grid = { # 'n_estimators': [20,30,40,50,75,80,100],\n                  \"max_depth\": [2,3,5,10,15,20,None],\n                  \"min_samples_split\": [2,5,7,10],\n                  \"min_samples_leaf\": [1,2,5] }  \n  clf = DecisionTreeClassifier(random_state=10)\n  grid_cv_DTC = RandomizedSearchCV(clf, param_grid, scoring=\"roc_auc\",cv=10).fit(X_train, y_train.values.ravel())\n  Evaluation( grid_cv_DTC,X_train,X_test,y_train,y_test )\n\n  # Random Forest\n  param_grid = { #'n_estimators': [50,75,100,200,500,1000],\n                  'max_features': [4,6,8,10,12,14,16,18],\n                  \"max_depth\": [2,3,5,10,15,20,None],\n                  \"min_samples_split\": [2,5,7,10],\n                  \"min_samples_leaf\": [1,2,5]  }\n  clf = RandomForestClassifier(random_state=10)\n  grid_cv_RFC = RandomizedSearchCV(clf, param_grid, scoring=\"roc_auc\",cv=10, verbose=0).fit(X_train, y_train.values.ravel())\n  Evaluation( grid_cv_RFC, X_train, X_test, y_train, y_test )\n \n  # XGBClassifier\n  param_grid = { \"max_depth\": [3, 4, 5, 7],\n                  \"learning_rate\": [0.1, 0.01, 0.05],\n                  \"gamma\": [0, 0.25, 1],\n                  \"reg_lambda\": [0, 0.5, 1, 5 ,10],\n                  \"scale_pos_weight\": [1, 3, 5],\n                  \"subsample\": [0.8],\n                  \"colsample_bytree\": [0.5] }\n  clf = XGBClassifier(random_state=10,verbose=0)\n  grid_cv_XGBC = RandomizedSearchCV(clf, param_grid, scoring=\"roc_auc\",cv=10).fit(X_train, y_train.values.ravel())\n  Evaluation( grid_cv_XGBC,X_train,X_test,y_train,y_test )\n\n  # Logistic Regression\n  param_grid = {'C': [0.1,0.5,1,10,50,100],\n                'max_iter': [100,250,500],\n                'fit_intercept':[True],\n                'intercept_scaling':[1],\n                'penalty':['l2'],\n                'tol':[0.00001,0.0001,0.000001]}\n  clf = LogisticRegression(random_state=10,verbose=0)\n  grid_cv_LR = RandomizedSearchCV(clf, param_grid, scoring=\"roc_auc\",cv=10).fit(X_train, y_train.values.ravel())\n  Evaluation( grid_cv_LR,X_train,X_test,y_train,y_test )\n\n  # Fit SVM with RBF Kernel\n  param_grid = {'C': [0.5,100,150],\n                'gamma': [0.1,0.01,0.001],\n                'probability':[True],\n                'kernel': ['rbf']}\n  grid_cv_SVC = RandomizedSearchCV(SVC(), param_grid, cv=10, refit=True, verbose=0)\n  grid_cv_SVC.fit(X_train, y_train )\n  Evaluation( grid_cv_SVC,X_train,X_test,y_train,y_test )\n  return\n","metadata":{"id":"ZBEPuqIarnzm","execution":{"iopub.status.busy":"2022-07-15T04:35:54.566669Z","iopub.execute_input":"2022-07-15T04:35:54.567564Z","iopub.status.idle":"2022-07-15T04:35:54.589044Z","shell.execute_reply.started":"2022-07-15T04:35:54.567529Z","shell.execute_reply":"2022-07-15T04:35:54.588165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"applying_hypertuning_models( X_train,X_test,y_train,y_test )","metadata":{"id":"CqAoGjCoO_50","outputId":"2fae1542-506b-4e0b-e563-ff6974851c64","execution":{"iopub.status.busy":"2022-07-15T04:35:54.590459Z","iopub.execute_input":"2022-07-15T04:35:54.591290Z","iopub.status.idle":"2022-07-15T04:41:37.364310Z","shell.execute_reply.started":"2022-07-15T04:35:54.591255Z","shell.execute_reply":"2022-07-15T04:41:37.363174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# -----------------------------------------------------------------------------------------------------------\n# RandomizedSearchCV(cv=10, estimator=XGBClassifier(random_state=10, verbose=0),\n#                    param_distributions={'colsample_bytree': [0.5],\n#                                         'gamma': [0, 0.25, 1],\n#                                         'learning_rate': [0.1, 0.01, 0.05],\n#                                         'max_depth': [3, 4, 5, 7],\n#                                         'reg_lambda': [0, 0.5, 1, 5, 10],\n#                                         'scale_pos_weight': [1, 3, 5],\n#                                         'subsample': [0.8]},\n#                    scoring='roc_auc')\n#  For Train Set :  \n# Param for GS {'subsample': 0.8, 'scale_pos_weight': 3, 'reg_lambda': 0, 'max_depth': 3, 'learning_rate': 0.1, 'gamma': 0.25, 'colsample_bytree': 0.5}\n# CV score for GS 0.9114758760717132\n# F1_Score =  0.9287411450393468\n# AUC ROC Score for GS:  0.9174993341523974\n#               precision    recall  f1-score   support\n\n#            0       0.98      0.99      0.98      2922\n#            1       0.91      0.85      0.88       478\n\n#     accuracy                           0.97      3400\n#    macro avg       0.94      0.92      0.93      3400\n# weighted avg       0.97      0.97      0.97      3400\n\n#  For Test Set :  \n# Test Set F1_Score =  0.9008873497277317\n# Test AUC ROC Score for GS:  0.8925228310502283\n#               precision    recall  f1-score   support\n\n#            0       0.97      0.98      0.97       730\n#            1       0.85      0.81      0.83       120\n\n#     accuracy                           0.95       850\n#    macro avg       0.91      0.89      0.90       850\n# weighted avg       0.95      0.95      0.95       850\n\n# ------------------------------------------------------------------------------------------------------------\n","metadata":{"id":"AldKrNjkRBh7","execution":{"iopub.status.busy":"2022-07-15T04:41:37.365758Z","iopub.execute_input":"2022-07-15T04:41:37.366060Z","iopub.status.idle":"2022-07-15T04:41:37.372374Z","shell.execute_reply.started":"2022-07-15T04:41:37.366033Z","shell.execute_reply":"2022-07-15T04:41:37.371333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* With hypertunner parameters, best performance is given by XGBClassifier and RandomForestClassifier but Overfitting in RandomForestClassifier is more as compared to XGBClassifier","metadata":{"id":"Bji5LUEzNjSQ"}},{"cell_type":"code","source":"px.imshow( X_train.corr(),color_continuous_scale='RdBu_r', width=700, height=700 )","metadata":{"id":"aaL6CNcd1Eaq","outputId":"acbd1c59-d140-4ea1-e70f-fd6dab1f650e","execution":{"iopub.status.busy":"2022-07-15T04:41:37.373798Z","iopub.execute_input":"2022-07-15T04:41:37.374077Z","iopub.status.idle":"2022-07-15T04:41:37.460862Z","shell.execute_reply.started":"2022-07-15T04:41:37.374050Z","shell.execute_reply":"2022-07-15T04:41:37.459961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape, y_train.shape, X_test.shape, y_test.shape","metadata":{"id":"NnmM_K5jZqPL","outputId":"3413901f-32fc-4319-a63b-24c6da44b2ee","execution":{"iopub.status.busy":"2022-07-15T04:41:37.462145Z","iopub.execute_input":"2022-07-15T04:41:37.462403Z","iopub.status.idle":"2022-07-15T04:41:37.468646Z","shell.execute_reply.started":"2022-07-15T04:41:37.462377Z","shell.execute_reply":"2022-07-15T04:41:37.468085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SMOTE","metadata":{"id":"RIS_Q-YG4rKW"}},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\nsmote = SMOTE( sampling_strategy='minority', random_state=10 )\nX_train_smote , y_train_smote = smote.fit_resample( X_train, y_train )","metadata":{"id":"nPeCJwYCX3rR","outputId":"39536ff3-fab0-4192-edce-da5c21a963ed","execution":{"iopub.status.busy":"2022-07-15T04:41:37.469330Z","iopub.execute_input":"2022-07-15T04:41:37.469544Z","iopub.status.idle":"2022-07-15T04:41:37.621025Z","shell.execute_reply.started":"2022-07-15T04:41:37.469523Z","shell.execute_reply":"2022-07-15T04:41:37.619892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.imshow( X_train_smote.corr(),color_continuous_scale='RdBu_r', width=700, height=700 )","metadata":{"id":"NgGR-jEN3uqW","outputId":"9bcecfbc-b780-4e17-d382-fd2bf9c67c22","execution":{"iopub.status.busy":"2022-07-15T04:41:37.622215Z","iopub.execute_input":"2022-07-15T04:41:37.622479Z","iopub.status.idle":"2022-07-15T04:41:37.678523Z","shell.execute_reply.started":"2022-07-15T04:41:37.622450Z","shell.execute_reply":"2022-07-15T04:41:37.677827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* ### No Significant changes in coorelation among features after applying SMOTE Oversampling\n","metadata":{"id":"gCEi2QNz310d"}},{"cell_type":"code","source":"X_train_smote.shape, y_train_smote.shape, X_test.shape, y_test.shape","metadata":{"id":"sYXhxniUcXrb","outputId":"18f64d63-dce8-4b40-b09c-c68327958edc","execution":{"iopub.status.busy":"2022-07-15T04:41:37.679794Z","iopub.execute_input":"2022-07-15T04:41:37.680285Z","iopub.status.idle":"2022-07-15T04:41:37.685807Z","shell.execute_reply.started":"2022-07-15T04:41:37.680256Z","shell.execute_reply":"2022-07-15T04:41:37.685168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\nfig.add_trace( go.Histogram( x = y_train['churn'] , name='y_train' ) )\nfig.add_trace( go.Histogram( x = y_train_smote['churn'] , name='y_train_smote' ) )\nfig.show()","metadata":{"id":"Ip5o51Di8gQV","outputId":"a4d6e1d1-bbd7-464c-a639-ae218c0ee6d4","execution":{"iopub.status.busy":"2022-07-15T04:41:37.689406Z","iopub.execute_input":"2022-07-15T04:41:37.691413Z","iopub.status.idle":"2022-07-15T04:41:37.706155Z","shell.execute_reply.started":"2022-07-15T04:41:37.691382Z","shell.execute_reply":"2022-07-15T04:41:37.705370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"apply_models_with_default_paramters(X_train_smote,X_test,y_train_smote,y_test)","metadata":{"id":"cVztcvb89Y7e","outputId":"cb01397f-ffd5-45fc-c3b4-ee59f6a85fc0","execution":{"iopub.status.busy":"2022-07-15T04:41:37.709776Z","iopub.execute_input":"2022-07-15T04:41:37.711530Z","iopub.status.idle":"2022-07-15T04:41:43.720527Z","shell.execute_reply.started":"2022-07-15T04:41:37.711498Z","shell.execute_reply":"2022-07-15T04:41:43.719729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* F1 Score for positive class doesn't increases ( in fact there is no change in any evaluestion reults ) ","metadata":{"id":"H4eeu_VTOJuq"}},{"cell_type":"code","source":"applying_hypertuning_models( X_train_smote,X_test,y_train_smote,y_test )","metadata":{"id":"l9FGimFzzwSZ","outputId":"08df0bdc-dc31-4d1a-fb90-101f9a9288e1","execution":{"iopub.status.busy":"2022-07-15T04:41:43.721531Z","iopub.execute_input":"2022-07-15T04:41:43.721786Z","iopub.status.idle":"2022-07-15T04:57:34.473480Z","shell.execute_reply.started":"2022-07-15T04:41:43.721760Z","shell.execute_reply":"2022-07-15T04:57:34.471944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 'Balanced' Ensembling Method","metadata":{"id":"xLVOfhGD5iMW"}},{"cell_type":"code","source":"from imblearn.ensemble import BalancedRandomForestClassifier\n\nparam_grid = { #'n_estimators': [50,75,100,200,500,1000],\n    \"max_depth\": [2,3,5,10,15,20,None],\n    \"min_samples_split\": [2,5,7,10],\n    \"min_samples_leaf\": [1,2,5] }\n\nbrfc = BalancedRandomForestClassifier()\ngrid_cv = RandomizedSearchCV(brfc, param_grid, scoring=\"roc_auc\",cv=10, verbose=0).fit(X_train, y_train.values.ravel())\n\nEvaluation(grid_cv,X_train,X_test,y_train,y_test,grid=True)\n\n# from sklearn.metrics import balanced_accuracy_score\n# bal_acc=balanced_accuracy_score(y_test,y_predictions)\n# print( 'Balanced Accuracy Score', bal_acc )","metadata":{"id":"ULBj7N4-Tm2Z","outputId":"9970a4c4-d2f9-4513-f7bb-e65f7a870345","execution":{"iopub.status.busy":"2022-07-15T04:57:34.475490Z","iopub.execute_input":"2022-07-15T04:57:34.475923Z","iopub.status.idle":"2022-07-15T04:58:15.829414Z","shell.execute_reply.started":"2022-07-15T04:57:34.475862Z","shell.execute_reply":"2022-07-15T04:58:15.828421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* BalancedRandomForestClassifier is not performing well as compared RandomForestClassifier and XGBClassifier","metadata":{"id":"xSYk-pRKJwOG"}},{"cell_type":"markdown","source":"# Final Model\n( Model with Highest F1_Score )","metadata":{"id":"HTdt6snBia20"}},{"cell_type":"code","source":"final_model = RandomForestClassifier(min_samples_split=5, min_samples_leaf=1, max_features=6, max_depth=15) # XGBClassifier(subsample=0.8, scale_pos_weight=3,reg_lambda=1, max_depth=4, learning_rate=0.1, gamma=1, colsample_bytree=0.5)\nfinal_model.fit( X_train, y_train )\n\n# Plotting ROC Curve\ny_pred = final_model.predict_proba( X_test )[:,1]\nfpr, tpr, thresholds = roc_curve( y_test, y_pred )\nauc_score = roc_auc_score( y_test, y_pred )\nfig = go.Figure()\nfig.add_trace( go.Scatter( x=fpr, y=tpr, name=\"AUC Score: \"+str(auc_score) ) )\n\n# roc curve for tpr = fpr \nrandom_probs = [0 for i in range(len(y_test))]\np_fpr, p_tpr, _ = roc_curve(y_test , random_probs, pos_label=1)\nfig.add_trace( go.Scatter(x=p_fpr, y=p_tpr, name='AUC Score: 0.5'))\nfig.update_layout( title='ROC Curve' )\nfig.show()","metadata":{"id":"SqKoqGNhicMr","outputId":"05a25850-fb79-42fb-bd9d-883443f77207","execution":{"iopub.status.busy":"2022-07-15T04:58:15.830381Z","iopub.execute_input":"2022-07-15T04:58:15.831142Z","iopub.status.idle":"2022-07-15T04:58:17.149086Z","shell.execute_reply.started":"2022-07-15T04:58:15.831109Z","shell.execute_reply":"2022-07-15T04:58:17.148169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_predictions = pd.DataFrame( final_model.predict(df_Test), columns=['churn'] )","metadata":{"id":"0BHm7EKz28Lm","outputId":"619d3ae7-e39a-4c30-87e0-3e9561a00991","execution":{"iopub.status.busy":"2022-07-15T04:58:17.150661Z","iopub.execute_input":"2022-07-15T04:58:17.151065Z","iopub.status.idle":"2022-07-15T04:58:17.180407Z","shell.execute_reply.started":"2022-07-15T04:58:17.151034Z","shell.execute_reply":"2022-07-15T04:58:17.179019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission_Data = pd.concat([Test_Data_identifier,final_predictions], axis='columns')\nSubmission_Data.churn.replace( [0,1],['no','yes'], inplace=True )\nSubmission_Data","metadata":{"id":"z7X9Rn9Pja6N","outputId":"4156c9a2-e962-4fa0-aa17-600d2923e9e0","execution":{"iopub.status.busy":"2022-07-15T04:58:17.181545Z","iopub.execute_input":"2022-07-15T04:58:17.181807Z","iopub.status.idle":"2022-07-15T04:58:17.196821Z","shell.execute_reply.started":"2022-07-15T04:58:17.181783Z","shell.execute_reply":"2022-07-15T04:58:17.195453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from google.colab import files\n# Submission_Data.to_csv('Customer_Churn_Prediction_Shirsh_Submission_File.csv', encoding = 'utf-8-sig', index=False) \n# files.download('Customer_Churn_Prediction_Shirsh_Submission_File.csv')","metadata":{"id":"uc8iMKbykXli","execution":{"iopub.status.busy":"2022-07-15T04:58:17.198467Z","iopub.execute_input":"2022-07-15T04:58:17.198714Z","iopub.status.idle":"2022-07-15T04:58:17.206860Z","shell.execute_reply.started":"2022-07-15T04:58:17.198692Z","shell.execute_reply":"2022-07-15T04:58:17.205957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#\nlist_objects_functions = { 'model':final_model, 'RS':RS, 'ohe':ohe }\n# save the model to disk\nimport pickle\nwith open('model.pkl', 'wb') as files:\n  pickle.dump(list_objects_functions, files)","metadata":{"id":"0UudNOUbs4eo","execution":{"iopub.status.busy":"2022-07-15T04:58:17.208766Z","iopub.execute_input":"2022-07-15T04:58:17.209075Z","iopub.status.idle":"2022-07-15T04:58:17.223812Z","shell.execute_reply.started":"2022-07-15T04:58:17.209045Z","shell.execute_reply":"2022-07-15T04:58:17.223146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pickle\n# import xgboost as xgb\n\n# dtrain = xgb.DMatrix(X_train, label=y_train)\n# bst = xgb.train({}, dtrain, 20)\n# bst.save_model('model.bst')","metadata":{"id":"_xLvC0Poxo-9","execution":{"iopub.status.busy":"2022-07-15T04:58:17.225292Z","iopub.execute_input":"2022-07-15T04:58:17.225620Z","iopub.status.idle":"2022-07-15T04:58:17.230839Z","shell.execute_reply.started":"2022-07-15T04:58:17.225588Z","shell.execute_reply":"2022-07-15T04:58:17.229834Z"},"trusted":true},"execution_count":null,"outputs":[]}]}