{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T19:21:28.579517Z","iopub.execute_input":"2022-07-21T19:21:28.580028Z","iopub.status.idle":"2022-07-21T19:21:28.616047Z","shell.execute_reply.started":"2022-07-21T19:21:28.579922Z","shell.execute_reply":"2022-07-21T19:21:28.615227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt #Used for data visulaization\nimport seaborn as sns #Used for data visualization\nimport warnings\nwarnings.filterwarnings('ignore') #To supress warnings","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:28.617814Z","iopub.execute_input":"2022-07-21T19:21:28.618327Z","iopub.status.idle":"2022-07-21T19:21:29.870734Z","shell.execute_reply.started":"2022-07-21T19:21:28.618290Z","shell.execute_reply":"2022-07-21T19:21:29.869733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/spaceship-titanic/train.csv\") #Loading dataset\ndf_train.head() #To see first five rows of the dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:29.872225Z","iopub.execute_input":"2022-07-21T19:21:29.873200Z","iopub.status.idle":"2022-07-21T19:21:29.975385Z","shell.execute_reply.started":"2022-07-21T19:21:29.873155Z","shell.execute_reply":"2022-07-21T19:21:29.974194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"../input/spaceship-titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:29.978910Z","iopub.execute_input":"2022-07-21T19:21:29.979394Z","iopub.status.idle":"2022-07-21T19:21:30.015820Z","shell.execute_reply.started":"2022-07-21T19:21:29.979351Z","shell.execute_reply":"2022-07-21T19:21:30.014484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"df.shape returns tuple of shape (rows, columns) of a series or a dataframe","metadata":{}},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:30.017262Z","iopub.execute_input":"2022-07-21T19:21:30.017619Z","iopub.status.idle":"2022-07-21T19:21:30.024310Z","shell.execute_reply.started":"2022-07-21T19:21:30.017576Z","shell.execute_reply":"2022-07-21T19:21:30.023428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To get a summary of a dataframe we use info() function. This method prints information about a dataframe including the index dtypes, column dtypes, non-null values and memory usage.","metadata":{}},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:30.025599Z","iopub.execute_input":"2022-07-21T19:21:30.026173Z","iopub.status.idle":"2022-07-21T19:21:30.070586Z","shell.execute_reply.started":"2022-07-21T19:21:30.026142Z","shell.execute_reply":"2022-07-21T19:21:30.068484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can verify the presence of null values using isnull() function.","metadata":{}},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:30.072477Z","iopub.execute_input":"2022-07-21T19:21:30.073241Z","iopub.status.idle":"2022-07-21T19:21:30.094681Z","shell.execute_reply.started":"2022-07-21T19:21:30.073191Z","shell.execute_reply":"2022-07-21T19:21:30.093106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Remove Duplicates if any**\n\nTo remove dupliactes from a dataframe, we can use pandas drop_duplicates() function. drop_duplicates() removes duplicate rows based on all columns.","metadata":{}},{"cell_type":"code","source":"data = df_train.drop_duplicates()\nprint(df_train.shape)\nprint(data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:30.098573Z","iopub.execute_input":"2022-07-21T19:21:30.102938Z","iopub.status.idle":"2022-07-21T19:21:30.138112Z","shell.execute_reply.started":"2022-07-21T19:21:30.102841Z","shell.execute_reply":"2022-07-21T19:21:30.136781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Unique HomePlanet:', df_train.HomePlanet.unique(), '\\nUnique Destination:', df_train.Destination.unique())","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:30.139403Z","iopub.execute_input":"2022-07-21T19:21:30.140240Z","iopub.status.idle":"2022-07-21T19:21:30.149741Z","shell.execute_reply.started":"2022-07-21T19:21:30.140197Z","shell.execute_reply":"2022-07-21T19:21:30.148518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the correlation\ncorr = df_train.corr()\n\n# Checking the correlation in heatmap\nplt.figure(figsize=(24,18))\n\nsns.heatmap(corr, cmap=\"coolwarm\", annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:30.155515Z","iopub.execute_input":"2022-07-21T19:21:30.156206Z","iopub.status.idle":"2022-07-21T19:21:30.884367Z","shell.execute_reply.started":"2022-07-21T19:21:30.156158Z","shell.execute_reply":"2022-07-21T19:21:30.882003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Fill the missing values**","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nimputer = SimpleImputer(missing_values=np.nan, strategy='most_frequent')\ndf_train = pd.DataFrame(imputer.fit_transform(df_train), columns=df_train.columns, index=df_train.index)\ndf_test = pd.DataFrame(imputer.fit_transform(df_test), columns=df_test.columns, index=df_test.index)\ndf_train = df_train.reset_index(drop=True)\ndf_test = df_test.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:30.885964Z","iopub.execute_input":"2022-07-21T19:21:30.886435Z","iopub.status.idle":"2022-07-21T19:21:31.390590Z","shell.execute_reply.started":"2022-07-21T19:21:30.886388Z","shell.execute_reply":"2022-07-21T19:21:31.389317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Visulaization**\nData visualization provides a good, organized pictorial representation of the data which makes it easier to understand, observe, analyze. It is useful for data cleaning, exploring data structure, detecting outliers and unusual groups, identifying trends and clusters, spotting local patterns, evaluating modeling output, and presenting results.","metadata":{}},{"cell_type":"code","source":"earth = df_train['HomePlanet'].value_counts()['Earth']\neuropa = df_train['HomePlanet'].value_counts()['Europa']\nmars = df_train['HomePlanet'].value_counts()['Mars']","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:31.392102Z","iopub.execute_input":"2022-07-21T19:21:31.392398Z","iopub.status.idle":"2022-07-21T19:21:31.404110Z","shell.execute_reply.started":"2022-07-21T19:21:31.392371Z","shell.execute_reply":"2022-07-21T19:21:31.402916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\nlabels=df_train.HomePlanet.unique()\nprint(labels)\nvalues= [europa, earth, mars]\nfig = px.pie(df_train, values=values, names=labels)\nfig.update_traces(textposition='inside', textinfo='percent+label')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:31.405654Z","iopub.execute_input":"2022-07-21T19:21:31.406591Z","iopub.status.idle":"2022-07-21T19:21:34.238769Z","shell.execute_reply.started":"2022-07-21T19:21:31.406499Z","shell.execute_reply":"2022-07-21T19:21:34.237613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Transported'].value_counts()[True]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:34.240268Z","iopub.execute_input":"2022-07-21T19:21:34.240731Z","iopub.status.idle":"2022-07-21T19:21:34.254883Z","shell.execute_reply.started":"2022-07-21T19:21:34.240697Z","shell.execute_reply":"2022-07-21T19:21:34.251700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Transported'].value_counts()[True]/df_train['Transported'].count()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:34.256701Z","iopub.execute_input":"2022-07-21T19:21:34.257394Z","iopub.status.idle":"2022-07-21T19:21:34.269237Z","shell.execute_reply.started":"2022-07-21T19:21:34.257358Z","shell.execute_reply":"2022-07-21T19:21:34.268069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transported = df_train['Transported'].value_counts()[True]\nnot_transported = df_train['Transported'].value_counts()[False]\ntrans_perc = round((transported/df_train['Transported'].count()*100),2)\nnot_trans_perc = round((not_transported/df_train['Transported'].count()*100),2)\ntravel_percentage = {'Travel':['Not Transported', 'Transported'], 'Percentage':[not_trans_perc, trans_perc]} \ndf_travel_percentage = pd.DataFrame(travel_percentage) \nsns.barplot(x='Travel',y='Percentage', data=df_travel_percentage)\nplt.title('Percentage of transported vs not transported')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:34.270929Z","iopub.execute_input":"2022-07-21T19:21:34.271687Z","iopub.status.idle":"2022-07-21T19:21:34.467405Z","shell.execute_reply.started":"2022-07-21T19:21:34.271640Z","shell.execute_reply":"2022-07-21T19:21:34.465937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,8))\nsns.histplot(df_train.Age)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:34.469652Z","iopub.execute_input":"2022-07-21T19:21:34.470193Z","iopub.status.idle":"2022-07-21T19:21:34.831457Z","shell.execute_reply.started":"2022-07-21T19:21:34.470142Z","shell.execute_reply":"2022-07-21T19:21:34.830119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.plot(kind='box',subplots=True,layout=(3,8),sharex=False,sharey=False,figsize=(20,10))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:34.833369Z","iopub.execute_input":"2022-07-21T19:21:34.833761Z","iopub.status.idle":"2022-07-21T19:21:36.033976Z","shell.execute_reply.started":"2022-07-21T19:21:34.833728Z","shell.execute_reply":"2022-07-21T19:21:36.032614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"passenger_id = df_test.PassengerId.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:36.035388Z","iopub.execute_input":"2022-07-21T19:21:36.035867Z","iopub.status.idle":"2022-07-21T19:21:36.040893Z","shell.execute_reply.started":"2022-07-21T19:21:36.035833Z","shell.execute_reply":"2022-07-21T19:21:36.039856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.Transported = df_train.Transported.astype('int')\ndf_train[['pass_grp','pass_no']]= df_train['PassengerId'].str.split('_', n = -1, expand = True)\ndf_train.drop('PassengerId', axis = 1, inplace = True)\n\ndf_test[['pass_grp','pass_no']]= df_test['PassengerId'].str.split('_', n = -1, expand = True)\ndf_test.drop('PassengerId', axis = 1, inplace = True)\n\n\ndf_train[['CryoSleep', 'VIP','Age','Spa','RoomService','FoodCourt','ShoppingMall','VRDeck','pass_grp','pass_no']] = df_train[['CryoSleep','VIP','Age','Spa','RoomService','FoodCourt','ShoppingMall','VRDeck','pass_grp','pass_no']].astype(int)\ndf_test[['CryoSleep', 'VIP','Age','Spa','RoomService','FoodCourt','ShoppingMall','VRDeck','pass_grp','pass_no']] = df_test[['CryoSleep','VIP','Age','Spa','RoomService','FoodCourt','ShoppingMall','VRDeck','pass_grp','pass_no']].astype(int)\n\ndf_train.drop(columns=['Cabin', 'Name'], inplace=True)\ndf_test.drop(columns=['Cabin', 'Name'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:36.042771Z","iopub.execute_input":"2022-07-21T19:21:36.043317Z","iopub.status.idle":"2022-07-21T19:21:36.128712Z","shell.execute_reply.started":"2022-07-21T19:21:36.043271Z","shell.execute_reply":"2022-07-21T19:21:36.127498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.get_dummies(df_train, columns=['HomePlanet', 'Destination'])\ndf_test = pd.get_dummies(df_test, columns=['HomePlanet', 'Destination'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:36.130323Z","iopub.execute_input":"2022-07-21T19:21:36.130670Z","iopub.status.idle":"2022-07-21T19:21:36.158774Z","shell.execute_reply.started":"2022-07-21T19:21:36.130639Z","shell.execute_reply":"2022-07-21T19:21:36.157699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **MODEL BUILDING**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split #Importing the train test split library\nX = df_train.drop(['Transported'], axis=1) # Putting feature variables into X\ny = df_train['Transported'] # Putting target variable to y\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=42, stratify = y) # Splitting data into train and test set 75:25","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:36.160752Z","iopub.execute_input":"2022-07-21T19:21:36.161189Z","iopub.status.idle":"2022-07-21T19:21:36.179341Z","shell.execute_reply.started":"2022-07-21T19:21:36.161147Z","shell.execute_reply":"2022-07-21T19:21:36.177919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import LogisticRegression,RidgeClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.ensemble import VotingClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:36.182863Z","iopub.execute_input":"2022-07-21T19:21:36.183334Z","iopub.status.idle":"2022-07-21T19:21:36.982451Z","shell.execute_reply.started":"2022-07-21T19:21:36.183299Z","shell.execute_reply":"2022-07-21T19:21:36.981009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifiers = {\n    \"Naive Bayes\": GaussianNB(),\n    \"Logistic Regression\":LogisticRegression(),\n    'Ridge Classifier':RidgeClassifier(),\n    \"Decision Tree\":DecisionTreeClassifier(),\n    \"Random Forest\":RandomForestClassifier(),\n    \"XG Boost\":XGBClassifier(),\n    \"Support Vector Machine\": SVC(),\n    \"K-Nearest Neighbors\": KNeighborsClassifier(),\n    'Cat Boost Classifier':CatBoostClassifier(),\n    'Gradient Boosting':GradientBoostingClassifier(),\n    \"Light GBM\": LGBMClassifier()\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:36.984220Z","iopub.execute_input":"2022-07-21T19:21:36.984896Z","iopub.status.idle":"2022-07-21T19:21:36.996792Z","shell.execute_reply.started":"2022-07-21T19:21:36.984856Z","shell.execute_reply":"2022-07-21T19:21:36.995406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import KFold, cross_val_score\nfrom sklearn import metrics\ncv = KFold(n_splits = 5,shuffle = True,random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:36.998379Z","iopub.execute_input":"2022-07-21T19:21:36.998841Z","iopub.status.idle":"2022-07-21T19:21:37.016972Z","shell.execute_reply.started":"2022-07-21T19:21:36.998803Z","shell.execute_reply":"2022-07-21T19:21:37.015776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Results = pd.DataFrame(columns=['Model','Accuracy','F1 score','ROC value','Score','CV-score'])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:37.018847Z","iopub.execute_input":"2022-07-21T19:21:37.019497Z","iopub.status.idle":"2022-07-21T19:21:37.029788Z","shell.execute_reply.started":"2022-07-21T19:21:37.019457Z","shell.execute_reply":"2022-07-21T19:21:37.028888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model(clf, df_Results, X_train,y_train, X_test, y_test):\n    for i, (clf_name,clf) in enumerate(classifiers.items()):\n        check  = cross_val_score(clf,X_train,y_train,cv = cv,scoring = \"accuracy\",n_jobs= -1)\n        clf.fit(X_train, y_train)\n        y_test_pred = clf.predict(X_test)\n\n        df_Results = df_Results.append(pd.DataFrame({'Model': clf_name,'Accuracy': metrics.accuracy_score(y_test, y_test_pred) ,'F1 score':metrics.f1_score(y_test, y_test_pred),'ROC value': metrics.roc_auc_score(y_test, y_test_pred),'Score':clf.score(X_test, y_test),'CV-score':check.mean()}, index=[0]),ignore_index= True)\n    return df_Results","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:37.031233Z","iopub.execute_input":"2022-07-21T19:21:37.031608Z","iopub.status.idle":"2022-07-21T19:21:37.042850Z","shell.execute_reply.started":"2022-07-21T19:21:37.031567Z","shell.execute_reply":"2022-07-21T19:21:37.041821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Results = model(classifiers,df_Results, X_train,y_train, X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:21:37.049239Z","iopub.execute_input":"2022-07-21T19:21:37.050137Z","iopub.status.idle":"2022-07-21T19:22:23.939054Z","shell.execute_reply.started":"2022-07-21T19:21:37.050092Z","shell.execute_reply":"2022-07-21T19:22:23.937838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Results","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:22:23.940985Z","iopub.execute_input":"2022-07-21T19:22:23.941334Z","iopub.status.idle":"2022-07-21T19:22:23.959461Z","shell.execute_reply.started":"2022-07-21T19:22:23.941302Z","shell.execute_reply":"2022-07-21T19:22:23.958491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_score = pd.DataFrame(data={'Model': df_Results['Model'], 'CV-score': df_Results['CV-score']})\n\nplt.figure(figsize=(20, 10))\nplot = sns.barplot(x=\"Model\", y=\"CV-score\", data=data_score, palette=\"magma\")\nplot.bar_label(plot.containers[0],fmt = \"%.3f\")\nplt.title('Performance analysis of different classifiers based on Score')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:22:23.960876Z","iopub.execute_input":"2022-07-21T19:22:23.961447Z","iopub.status.idle":"2022-07-21T19:22:24.310620Z","shell.execute_reply.started":"2022-07-21T19:22:23.961416Z","shell.execute_reply":"2022-07-21T19:22:24.309542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_accuracy = pd.DataFrame(data={'Model': df_Results['Model'], 'Accuracy': df_Results['Accuracy']})\n\nplt.figure(figsize=(20, 10))\nplot = sns.barplot(x=\"Model\", y=\"Accuracy\", data=data_accuracy, palette=\"magma\")\nplot.bar_label(plot.containers[0],fmt = \"%.3f\")\nplt.title('Performance analysis of different classifiers based on Accuracy')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:22:24.312046Z","iopub.execute_input":"2022-07-21T19:22:24.312660Z","iopub.status.idle":"2022-07-21T19:22:24.653422Z","shell.execute_reply.started":"2022-07-21T19:22:24.312602Z","shell.execute_reply":"2022-07-21T19:22:24.652460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cat Boost, Gradient Boosting and Light GBM. Let us tune them and check","metadata":{}},{"cell_type":"markdown","source":"## **MODEL TUNING USING PIPELINES**","metadata":{}},{"cell_type":"code","source":"models_tune = {\n    \"CatBoost\": CatBoostClassifier(random_state=42,verbose = 0),\n    \"XGB\": XGBClassifier(random_state=42),\n    \"LGBM\": LGBMClassifier(random_state=42)\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:22:24.654870Z","iopub.execute_input":"2022-07-21T19:22:24.655485Z","iopub.status.idle":"2022-07-21T19:22:24.661651Z","shell.execute_reply.started":"2022-07-21T19:22:24.655450Z","shell.execute_reply":"2022-07-21T19:22:24.660245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params_catboost= {\n    \"model__learning_rate\":(0.001,1),\n    \"model__max_depth\":(1,7),\n    \"model__l2_leaf_reg\":(0.001,100)\n}\n\nparams_xgb = {\n    \"model__gamma\":(1,10),\n    \"model__learning_rate\":(0.01,1),\n    \"model__max_depth\":(1,15),\n    \"model__reg_alpha\":(0.001,100),\n    \"model__reg_lambda\":(0.001,100)\n    \n}\n\nparams_lgbm = {\n    \"model__max_depth\":(3,12),\n    \"model__learning_rate\":(0.001,1),\n    \"model__subsample\":(0.1,1),\n    \"model__reg_alpha\":(0.001,100),\n    \"model__colsample_bytree\":(0.1,1),\n    \"model__reg_lambda\":(0.001,100)\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:22:24.663781Z","iopub.execute_input":"2022-07-21T19:22:24.664362Z","iopub.status.idle":"2022-07-21T19:22:24.677573Z","shell.execute_reply.started":"2022-07-21T19:22:24.664310Z","shell.execute_reply":"2022-07-21T19:22:24.676126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_params = dict(zip(models_tune, [params_catboost, params_lgbm, params_xgb]))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:22:24.679323Z","iopub.execute_input":"2022-07-21T19:22:24.679776Z","iopub.status.idle":"2022-07-21T19:22:24.691869Z","shell.execute_reply.started":"2022-07-21T19:22:24.679739Z","shell.execute_reply":"2022-07-21T19:22:24.690616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_dict = {}","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:22:24.694043Z","iopub.execute_input":"2022-07-21T19:22:24.694448Z","iopub.status.idle":"2022-07-21T19:22:24.704228Z","shell.execute_reply.started":"2022-07-21T19:22:24.694402Z","shell.execute_reply":"2022-07-21T19:22:24.702861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import RandomizedSearchCV\n\nfor model in models_tune:\n    models = Pipeline([(\"model\",models_tune[model])]) # list(models_tune.values())[i])]) since dict is unhashable I didn't use this.\n    model_dict[model] = RandomizedSearchCV(models,models_params[model], cv=cv, scoring='accuracy', n_iter=20, n_jobs=-1, verbose=0, random_state=42)\n    model_dict[model].fit(X_train, y_train)\n    print(model)\n    print(\"Best parameters found on training set:\")\n    print(model_dict[model].best_params_)\n    print(\"Best score found on training set, validation set, and test set :\")\n    print(model_dict[model].score(X_train, y_train), model_dict[model].best_score_, model_dict[model].score(X_test, y_test))\n    print('---'*60)\n    print('\\n')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:22:24.705681Z","iopub.execute_input":"2022-07-21T19:22:24.706538Z","iopub.status.idle":"2022-07-21T19:25:05.745838Z","shell.execute_reply.started":"2022-07-21T19:22:24.706504Z","shell.execute_reply":"2022-07-21T19:25:05.744650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"XGB seems to perform better. \n\nWe can also tune without pipelines (since there will be repeatition in code, I used this method).","metadata":{}},{"cell_type":"code","source":"y_pred = model_dict[\"XGB\"].predict(df_test)\ny_pred = (y_pred == 1)\nsubmission = pd.DataFrame(columns=[\"PassengerId\",\"Transported\"])\nsubmission[\"PassengerId\"] = passenger_id\nsubmission['Transported'] = y_pred\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:25:05.747519Z","iopub.execute_input":"2022-07-21T19:25:05.748153Z","iopub.status.idle":"2022-07-21T19:25:05.789906Z","shell.execute_reply.started":"2022-07-21T19:25:05.748114Z","shell.execute_reply":"2022-07-21T19:25:05.788870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:25:05.791729Z","iopub.execute_input":"2022-07-21T19:25:05.792055Z","iopub.status.idle":"2022-07-21T19:25:05.810218Z","shell.execute_reply.started":"2022-07-21T19:25:05.792026Z","shell.execute_reply":"2022-07-21T19:25:05.808846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from subprocess import check_output\n# print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n# print(check_output([\"ls\", \"../working\"]).decode(\"utf8\"))\n# from IPython.display import FileLink\n# FileLink(r'submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:25:05.813728Z","iopub.execute_input":"2022-07-21T19:25:05.815003Z","iopub.status.idle":"2022-07-21T19:25:05.820307Z","shell.execute_reply.started":"2022-07-21T19:25:05.814954Z","shell.execute_reply":"2022-07-21T19:25:05.819200Z"},"trusted":true},"execution_count":null,"outputs":[]}]}