{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Predict Potential Spammers on Fiverr**","metadata":{}},{"cell_type":"markdown","source":"### Import Required Libraries","metadata":{}},{"cell_type":"code","source":"# Data Processing Libraries\nimport pandas as pd\nimport numpy as np\n# Visualization Libraries\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Import required libraries for model building\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBClassifier\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:00.032847Z","iopub.execute_input":"2022-08-12T12:55:00.033369Z","iopub.status.idle":"2022-08-12T12:55:01.691538Z","shell.execute_reply.started":"2022-08-12T12:55:00.033260Z","shell.execute_reply":"2022-08-12T12:55:01.689499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load Datasets","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('../input/predict-potential-spammers-on-fiverr/train.csv')\ndf_test = pd.read_csv('../input/predict-potential-spammers-on-fiverr/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:01.694807Z","iopub.execute_input":"2022-08-12T12:55:01.695289Z","iopub.status.idle":"2022-08-12T12:55:04.361526Z","shell.execute_reply.started":"2022-08-12T12:55:01.695249Z","shell.execute_reply":"2022-08-12T12:55:04.360221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check the shape of datasets\nprint('Train Data Shape: ', df_train.shape)\nprint('Test Data Shape: ', df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:04.365865Z","iopub.execute_input":"2022-08-12T12:55:04.366346Z","iopub.status.idle":"2022-08-12T12:55:04.372880Z","shell.execute_reply.started":"2022-08-12T12:55:04.366308Z","shell.execute_reply":"2022-08-12T12:55:04.371348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check the glimpse of first five rows of train data\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:04.376244Z","iopub.execute_input":"2022-08-12T12:55:04.377691Z","iopub.status.idle":"2022-08-12T12:55:04.417244Z","shell.execute_reply.started":"2022-08-12T12:55:04.377637Z","shell.execute_reply":"2022-08-12T12:55:04.415970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Summary Statistics \ndf_train.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:04.418751Z","iopub.execute_input":"2022-08-12T12:55:04.419151Z","iopub.status.idle":"2022-08-12T12:55:05.205663Z","shell.execute_reply.started":"2022-08-12T12:55:04.419114Z","shell.execute_reply":"2022-08-12T12:55:05.204452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets get an overview of features datatype\ndf_train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:05.207068Z","iopub.execute_input":"2022-08-12T12:55:05.207832Z","iopub.status.idle":"2022-08-12T12:55:05.218658Z","shell.execute_reply.started":"2022-08-12T12:55:05.207794Z","shell.execute_reply":"2022-08-12T12:55:05.217196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# checking missing values \ndf_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:05.220484Z","iopub.execute_input":"2022-08-12T12:55:05.221265Z","iopub.status.idle":"2022-08-12T12:55:05.269856Z","shell.execute_reply.started":"2022-08-12T12:55:05.221214Z","shell.execute_reply":"2022-08-12T12:55:05.267875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Null Values in X13**","metadata":{}},{"cell_type":"code","source":"# lets check the value count for missing values feature for data cleaning\ndf_train.X13.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:05.271498Z","iopub.execute_input":"2022-08-12T12:55:05.272411Z","iopub.status.idle":"2022-08-12T12:55:05.288971Z","shell.execute_reply.started":"2022-08-12T12:55:05.272359Z","shell.execute_reply":"2022-08-12T12:55:05.287747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count plot of target variable\nplt.figure(figsize=(12,6))\nsns.countplot(data=df_train, x = 'label')\nplt.title('Count plot of Label')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:05.290588Z","iopub.execute_input":"2022-08-12T12:55:05.292221Z","iopub.status.idle":"2022-08-12T12:55:05.529369Z","shell.execute_reply.started":"2022-08-12T12:55:05.292177Z","shell.execute_reply":"2022-08-12T12:55:05.528383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the unique values in each feature\ndf_train.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:05.533731Z","iopub.execute_input":"2022-08-12T12:55:05.534223Z","iopub.status.idle":"2022-08-12T12:55:05.714082Z","shell.execute_reply.started":"2022-08-12T12:55:05.534176Z","shell.execute_reply":"2022-08-12T12:55:05.712541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are features with only 1 unique value. These does not contribute to the model performance. Therefore these column will be removed from the dataset.","metadata":{}},{"cell_type":"code","source":"# features with only one unique values\ntrain_features_to_drop = df_train.columns[df_train.nunique() < 2]\ntrain_features_to_drop","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:05.717403Z","iopub.execute_input":"2022-08-12T12:55:05.717918Z","iopub.status.idle":"2022-08-12T12:55:05.896235Z","shell.execute_reply.started":"2022-08-12T12:55:05.717862Z","shell.execute_reply":"2022-08-12T12:55:05.894907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop irrelevant features \ndf_train_new = df_train.drop(train_features_to_drop,axis=1)\ndf_train_new.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:05.898374Z","iopub.execute_input":"2022-08-12T12:55:05.898943Z","iopub.status.idle":"2022-08-12T12:55:05.964358Z","shell.execute_reply.started":"2022-08-12T12:55:05.898888Z","shell.execute_reply":"2022-08-12T12:55:05.963065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# correlation matrix\ncor = df_train_new.corr()\n# correlation matrix heatmap\nplt.figure(figsize=(30,30))\nsns.heatmap(cor, annot = True, cmap=\"plasma\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:05.965999Z","iopub.execute_input":"2022-08-12T12:55:05.967195Z","iopub.status.idle":"2022-08-12T12:55:17.684400Z","shell.execute_reply.started":"2022-08-12T12:55:05.967144Z","shell.execute_reply":"2022-08-12T12:55:17.683115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets extract upper part only to see correlated features\nupper_correlation = cor.where(np.triu(np.ones(cor.shape),k=1).astype(bool))\nhighly_correlated_features = [col for col in upper_correlation.columns if any(upper_correlation[col] > 0.8)]\nprint(highly_correlated_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:17.685853Z","iopub.execute_input":"2022-08-12T12:55:17.686206Z","iopub.status.idle":"2022-08-12T12:55:17.705522Z","shell.execute_reply.started":"2022-08-12T12:55:17.686173Z","shell.execute_reply":"2022-08-12T12:55:17.704604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop highly correlated features\ndf_train_new = df_train_new.drop(highly_correlated_features, axis=1)\ndf_train_new.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:17.707115Z","iopub.execute_input":"2022-08-12T12:55:17.707761Z","iopub.status.idle":"2022-08-12T12:55:17.765100Z","shell.execute_reply.started":"2022-08-12T12:55:17.707723Z","shell.execute_reply":"2022-08-12T12:55:17.763752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Data Prepocessing on Test data\n\n# Drop 1 unique value features from test data\nonly_unique_value_test_features = df_test.columns[df_test.nunique() < 2]\n\n# Drop highly correlated features and user_id from test data\ntest_features_to_drop_sublist = [only_unique_value_test_features, highly_correlated_features, ['user_id']]\ntest_features_to_drop = [item for sublist in test_features_to_drop_sublist for item in sublist]\ndf_test_new = df_test.drop(test_features_to_drop,axis=1)\nprint(df_test_new.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:17.767171Z","iopub.execute_input":"2022-08-12T12:55:17.767944Z","iopub.status.idle":"2022-08-12T12:55:17.796084Z","shell.execute_reply.started":"2022-08-12T12:55:17.767885Z","shell.execute_reply":"2022-08-12T12:55:17.794746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Building","metadata":{}},{"cell_type":"code","source":"# Split traindf into dependent(y) and indepedent variables(X)\nX = df_train_new.drop(['label','user_id'],axis=1)\ny = df_train_new['label']","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:17.798020Z","iopub.execute_input":"2022-08-12T12:55:17.798905Z","iopub.status.idle":"2022-08-12T12:55:17.854103Z","shell.execute_reply.started":"2022-08-12T12:55:17.798838Z","shell.execute_reply":"2022-08-12T12:55:17.852763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data into training and test sets\n# X_train, X_test, y_train, y_test = train_test_split(X, y,random_state=2022)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:17.856290Z","iopub.execute_input":"2022-08-12T12:55:17.857092Z","iopub.status.idle":"2022-08-12T12:55:17.862236Z","shell.execute_reply.started":"2022-08-12T12:55:17.857013Z","shell.execute_reply":"2022-08-12T12:55:17.860799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting up a pipeline with median as an imputer and StandardScaler so that the features have 0 mean and a variance of 1 \n# then classifier - Baseline Model\nbase_pipeline = Pipeline(\n    steps=[(\"imputer\", SimpleImputer(strategy=\"median\")), \n           (\"scale\", StandardScaler()),\n          ('base_xgb', XGBClassifier())]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:17.863895Z","iopub.execute_input":"2022-08-12T12:55:17.864488Z","iopub.status.idle":"2022-08-12T12:55:17.875604Z","shell.execute_reply.started":"2022-08-12T12:55:17.864437Z","shell.execute_reply":"2022-08-12T12:55:17.874138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fitting pipeline to training data\nbase_pipeline.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:55:17.877288Z","iopub.execute_input":"2022-08-12T12:55:17.878090Z","iopub.status.idle":"2022-08-12T12:56:11.906020Z","shell.execute_reply.started":"2022-08-12T12:55:17.878014Z","shell.execute_reply":"2022-08-12T12:56:11.904766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction on Test Data","metadata":{}},{"cell_type":"code","source":"# Test Prediction\nbase_pred = base_pipeline.predict(df_test_new)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:56:11.907597Z","iopub.execute_input":"2022-08-12T12:56:11.908024Z","iopub.status.idle":"2022-08-12T12:56:11.961994Z","shell.execute_reply.started":"2022-08-12T12:56:11.907986Z","shell.execute_reply":"2022-08-12T12:56:11.960831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a submission df to test the base model result\nuser_id = df_test['user_id']\n\nbase_submission = pd.DataFrame({\n    'user_id': user_id,\n    'prediction': base_pred\n})\n\nprint(base_submission.prediction.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:56:11.963614Z","iopub.execute_input":"2022-08-12T12:56:11.964356Z","iopub.status.idle":"2022-08-12T12:56:11.975821Z","shell.execute_reply.started":"2022-08-12T12:56:11.964303Z","shell.execute_reply":"2022-08-12T12:56:11.974606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### HyperParameter Tuning","metadata":{}},{"cell_type":"code","source":"# Setting up a pipeline with imputer, scaler and classifier for hyperparameter tuning\ntuning_pipeline = Pipeline(\n    steps=[(\"imputer\", SimpleImputer(strategy=\"median\")), \n           (\"scaler\", StandardScaler()),\n          ('xgb', XGBClassifier())]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:57:57.014428Z","iopub.execute_input":"2022-08-12T12:57:57.014863Z","iopub.status.idle":"2022-08-12T12:57:57.021746Z","shell.execute_reply.started":"2022-08-12T12:57:57.014823Z","shell.execute_reply":"2022-08-12T12:57:57.020478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Putting together a parameter grid to search over using grid search\n# Parameters of pipelines can be set using '__' separated parameter names:\n# param_grid = {\n#     \"xgb__max_depth\": [5, 7,8,9,10],\n#     \"xgb__n_estimators\": np.arange(100,300,50),\n#     \"xgb__learning_rate\": [ 0.05,0.1,0.15, 0.23,],\n#     \"xgb__gamma\": [0.1, 0.25],\n#     \"xgb__reg_lambda\" : [0.95],\n#     \"xgb__reg_alpha\" : [0.35]\n# }","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:58:30.841389Z","iopub.execute_input":"2022-08-12T12:58:30.842910Z","iopub.status.idle":"2022-08-12T12:58:30.848145Z","shell.execute_reply.started":"2022-08-12T12:58:30.842856Z","shell.execute_reply":"2022-08-12T12:58:30.847006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setting up the grid search\n# grid = GridSearchCV(tuning_pipeline, param_grid, n_jobs=-1,cv=3,scoring='f1_macro',verbose=4)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:59:08.393009Z","iopub.execute_input":"2022-08-12T12:59:08.393451Z","iopub.status.idle":"2022-08-12T12:59:08.398433Z","shell.execute_reply.started":"2022-08-12T12:59:08.393416Z","shell.execute_reply":"2022-08-12T12:59:08.397258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fitting grid to training data\n# grid.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:59:37.281848Z","iopub.execute_input":"2022-08-12T12:59:37.282365Z","iopub.status.idle":"2022-08-12T12:59:37.287818Z","shell.execute_reply.started":"2022-08-12T12:59:37.282289Z","shell.execute_reply":"2022-08-12T12:59:37.286635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save the model\n# import pickle\n# pickle.dump(grid,open('xgbtune_grid_search.pkl','wb'))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:59:50.551834Z","iopub.execute_input":"2022-08-12T12:59:50.552283Z","iopub.status.idle":"2022-08-12T12:59:50.558684Z","shell.execute_reply.started":"2022-08-12T12:59:50.552244Z","shell.execute_reply":"2022-08-12T12:59:50.557137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's create a final pipeline with the grid selected parameters:\npipeline = Pipeline(\n    steps=[(\"imputer\", SimpleImputer(strategy=\"median\")), \n           (\"scaler\", StandardScaler()),\n          ('xgb', XGBClassifier(max_depth = 8,\n                                learning_rate=0.22,\n                                n_estimators=250,\n                                reg_alpha = 0.35,\n                                reg_lambda = 0.95,\n                                gamma = 0.1))]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:06:17.783191Z","iopub.execute_input":"2022-08-12T13:06:17.783702Z","iopub.status.idle":"2022-08-12T13:06:17.790749Z","shell.execute_reply.started":"2022-08-12T13:06:17.783663Z","shell.execute_reply":"2022-08-12T13:06:17.789463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fitting final classifier to training data\npipeline.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:06:30.521474Z","iopub.execute_input":"2022-08-12T13:06:30.521972Z","iopub.status.idle":"2022-08-12T13:09:35.003721Z","shell.execute_reply.started":"2022-08-12T13:06:30.521932Z","shell.execute_reply":"2022-08-12T13:09:35.002336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prediction on Test Data\npred = pipeline.predict(df_test_new)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:09:50.831524Z","iopub.execute_input":"2022-08-12T13:09:50.831949Z","iopub.status.idle":"2022-08-12T13:09:50.945641Z","shell.execute_reply.started":"2022-08-12T13:09:50.831912Z","shell.execute_reply":"2022-08-12T13:09:50.943840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission dataframe\nsubmission = pd.DataFrame({\n    'user_id': user_id,\n    'prediction': pred\n})","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:09:52.210756Z","iopub.execute_input":"2022-08-12T13:09:52.211237Z","iopub.status.idle":"2022-08-12T13:09:52.217858Z","shell.execute_reply.started":"2022-08-12T13:09:52.211201Z","shell.execute_reply":"2022-08-12T13:09:52.216462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.prediction.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:09:53.880958Z","iopub.execute_input":"2022-08-12T13:09:53.881381Z","iopub.status.idle":"2022-08-12T13:09:53.891393Z","shell.execute_reply.started":"2022-08-12T13:09:53.881345Z","shell.execute_reply":"2022-08-12T13:09:53.890135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission.to_csv('xgb_tune_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:10:08.169877Z","iopub.execute_input":"2022-08-12T13:10:08.170319Z","iopub.status.idle":"2022-08-12T13:10:08.175769Z","shell.execute_reply.started":"2022-08-12T13:10:08.170283Z","shell.execute_reply":"2022-08-12T13:10:08.174454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot feature importance\nfrom xgboost import plot_importance\nmodel = pipeline.named_steps[\"xgb\"]\n\nplt.figure(figsize=(15,10))    \nplot_importance(model)\nplt.rcParams[\"figure.figsize\"] = (10, 25)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:11:42.031988Z","iopub.execute_input":"2022-08-12T13:11:42.033291Z","iopub.status.idle":"2022-08-12T13:11:42.642396Z","shell.execute_reply.started":"2022-08-12T13:11:42.033243Z","shell.execute_reply":"2022-08-12T13:11:42.641299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check threshold value\nthresholds = np.sort(model.feature_importances_)\nprint(thresholds)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:12:47.282128Z","iopub.execute_input":"2022-08-12T13:12:47.282644Z","iopub.status.idle":"2022-08-12T13:12:47.292931Z","shell.execute_reply.started":"2022-08-12T13:12:47.282604Z","shell.execute_reply":"2022-08-12T13:12:47.291603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import SelectFromModel\nthresh = 0.00670489\n# select features using threshold\nselection = SelectFromModel(model, threshold=thresh, prefit=True)\nselect_X = selection.transform(X)\nprint('Train Data Shape: ', select_X.shape)\nselect_test = selection.transform(df_test_new)\nprint('Test Data Shape: ', select_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:16:39.939972Z","iopub.execute_input":"2022-08-12T13:16:39.940515Z","iopub.status.idle":"2022-08-12T13:16:40.054966Z","shell.execute_reply.started":"2022-08-12T13:16:39.940472Z","shell.execute_reply":"2022-08-12T13:16:40.052804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit model on selected features\npipeline.fit(select_X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:17:00.901072Z","iopub.execute_input":"2022-08-12T13:17:00.901513Z","iopub.status.idle":"2022-08-12T13:19:25.652285Z","shell.execute_reply.started":"2022-08-12T13:17:00.901478Z","shell.execute_reply":"2022-08-12T13:19:25.651101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction on test data\nfinal_pred = pipeline.predict(select_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:19:25.654710Z","iopub.execute_input":"2022-08-12T13:19:25.656379Z","iopub.status.idle":"2022-08-12T13:19:25.766484Z","shell.execute_reply.started":"2022-08-12T13:19:25.656325Z","shell.execute_reply":"2022-08-12T13:19:25.765464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_submission = pd.DataFrame({\n    'user_id': user_id,\n    'prediction': final_pred\n})\n\nprint(final_submission.prediction.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:19:57.360200Z","iopub.execute_input":"2022-08-12T13:19:57.360771Z","iopub.status.idle":"2022-08-12T13:19:57.370807Z","shell.execute_reply.started":"2022-08-12T13:19:57.360723Z","shell.execute_reply":"2022-08-12T13:19:57.369683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save the result into csv file\nfinal_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:20:37.512075Z","iopub.execute_input":"2022-08-12T13:20:37.512552Z","iopub.status.idle":"2022-08-12T13:20:37.549725Z","shell.execute_reply.started":"2022-08-12T13:20:37.512513Z","shell.execute_reply":"2022-08-12T13:20:37.548639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}