{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import packages\n\nimport pandas as pd, numpy as np, gc, joblib\n\nfrom xgboost import XGBClassifier\n\n#from sklearn.model_selection import RandomizedSearchCV\n#from sklearn.model_selection import GridSearchCV\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-24T18:58:17.794624Z","iopub.execute_input":"2022-08-24T18:58:17.795034Z","iopub.status.idle":"2022-08-24T18:58:19.608351Z","shell.execute_reply.started":"2022-08-24T18:58:17.79495Z","shell.execute_reply":"2022-08-24T18:58:19.607058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/amex-default-prediction/train_data.csv')\n\nprint(train.info())\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:59:00.862097Z","iopub.execute_input":"2022-08-24T18:59:00.862439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.S_2 = pd.to_datetime(train.S_2)\n\ntrain.S_2 = pd.to_numeric(train.S_2)\n\ncategorical_cols_list = list(train.select_dtypes(include = ['object', 'category']).columns)\nnumeric_cols_list = list(train.select_dtypes(include = ['float16', 'float64', 'int64']).columns)\n\n#categorical_cols_list = list(categorical_cols)\nfor col in categorical_cols_list: \n    if train[col].isnull().sum() == 0:\n        categorical_cols_list.remove(col)\n        \n\n#numeric_cols_list = list(numeric_cols)\nfor col in numeric_cols_list: \n    if train[col].isnull().sum() == 0:\n        numeric_cols_list.remove(col)\n\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_categorical = SimpleImputer(missing_values = np.nan, strategy = 'most_frequent')\n\ntrain[categorical_cols_list]=imp_categorical.fit_transform(train[categorical_cols_list])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#imp_numeric     = SimpleImputer(missing_values = np.nan, strategy = 'median')\n#train[numeric_cols_list] = imp_numeric.fit_transform(train[numeric_cols_list])\n# this doesn't work: not enough memory for the operation\n\n# replace null values with zeros\n\ntrain[numeric_cols_list] = train[numeric_cols_list].replace(np.nan, 0) #if this doesnt work try another value for nulls than np.nan\n\n#replace with modes instead: \n\n#train.fillna(modes, inplace = True)\n\ntrain.isna().sum().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_cols = list(train.select_dtypes(include = ['object', 'category']).columns)\ncategorical_cols.remove('customer_ID')\n\nnumeric_cols = train.select_dtypes(include = ['float16', 'float64', 'int64']).columns\n\nfor col in train[categorical_cols]:\n    train[col] = train[col].astype('category')\n\nfor col in train[categorical_cols]:\n    train[col] = train[col].cat.codes\n\n    \nscaler = StandardScaler()\n\n#scaler.fit_transform(train[numeric_cols])  do we have enough memory for this to work? =======================================================\n\n# this doesn't work: not enough memory. use standard scaler on each feature, one at \n# a time\n\nfor col in train[numeric_cols]:\n    train[col] = scaler.fit_transform(train[col].values.reshape(-1, 1))\n\n# group by Customer ID and then drop Customer ID: Unique values: no predictive value\ntrain = train.groupby(['customer_ID']).nth(-1).reset_index(drop=True)\n\nprint(len(train))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\n\nx_train = train\n\ny_train = targets['target']\n\n# delete the train dataframe to free memory\n\ndel train\ndel targets\ngc.collect()\ntrain = pd.DataFrame()\ntargets = pd.DataFrame()\n\nprint('Features & Label defined successfully')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model = XGBClassifier(\n learning_rate =0.01,\n n_estimators=5000, \n max_depth=4,\n min_child_weight=6,\n gamma=0,\n subsample=0.8,\n colsample_bytree=0.8,\n reg_alpha=0.005,\n objective= 'binary:logistic',\n nthread=4,\n scale_pos_weight=1,\n seed=27)\n\n\nxgb_model2 = XGBClassifier(\n learning_rate =0.01,\n n_estimators=10000,\n max_depth=12,\n min_child_weight=6,\n gamma=0,\n subsample=0.8,\n colsample_bytree=0.8,\n reg_alpha=0.005,\n objective= 'binary:logistic',\n nthread=4,\n scale_pos_weight=1,\n seed=0)\n\nxgb_model3 = XGBClassifier(\n learning_rate =0.01,\n n_estimators=13000,\n max_depth=12,\n min_child_weight=6,\n gamma=0,\n subsample=0.8,\n colsample_bytree=0.8,\n reg_alpha=0.005,\n objective= 'binary:logistic',\n nthread=4,\n scale_pos_weight=1,\n seed=0)\n\n# Hyperparameter tuning. Many thanks to the authors of these posts:\n# https://towardsdatascience.com/binary-classification-xgboost-hyperparameter-tuning-scenarios-by-non-exhaustive-grid-search-and-c261f4ce098d\n# https://towardsdatascience.com/doing-xgboost-hyper-parameter-tuning-the-smart-way-part-1-of-2-f6d255a45dde\n\n\n#parameters = {\"learning_rate\"    : [0.05, 0.10, 0.15, 0.20, 0.25, 0.30 ] ,\n              #\"max_depth\"        : [ 3, 4, 5, 6, 8, 10, 12, 15],\n              #\"min_child_weight\" : [ 1, 3, 5, 7 ],\n              #\"gamma\"            : [ 0.0, 0.1, 0.2 , 0.3, 0.4 ],\n              #\"colsample_bytree\" : [ 0.3, 0.4, 0.5 , 0.7 ] }\n\n#tuned_xgb = GridSearchCV(estimator=xgb_model, scoring='accuracy', \n                         #param_grid=parameters, \n                         #return_train_score=True,  verbose=1, cv=3)\n\n%time xgb_model.fit(x_train, y_train)\njoblib.dump(xgb_model, 'xgb model.pkl')\nprint('Xgb model fit successfully')\n\n%time xgb_model2.fit(x_train, y_train)\njoblib.dump(xgb_model2, 'xgb model2.pkl')\nprint('Xgb model2 fit successfully')\n\n%time xgb_model3.fit(x_train, y_train)\njoblib.dump(xgb_model3, 'xgb model3.pkl')\nprint('Xgb model3 fit successfully')\n\n#tuned_xgb.fit(x_train, y_train)\n#xgb_predictions = tuned_xgb.best_estimator_.predict_proba(x_test)\n#xgb_feature_importances = xgb_model.booster().get_fscore().sort_values(ascending=False)\n#joblib.dump(tuned_xgb, 'tuned xgb model.pkl')\n\nprint('All done')\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del x_train\ndel y_train\n\ndel xgb_model\ndel xgb_model2\ndel xgb_model3\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test = pd.read_csv(\"..\\test_data.csv\", dtype = dict_for_df)\n\n#test.to_pickle(r\"..\\test_data.pickle\")\n#test.to_feather(\"..\\test_data.ftr\")\n\ntest = pd.read_pickle('../input/amex-default-prediction/test_data.csv')\n\nprint(test.info())\n\ntest.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.S_2 = pd.to_datetime(test.S_2)\n\ntest.S_2 = pd.to_numeric(test.S_2)\n\ncategorical_cols_list = list(test.select_dtypes(include = ['object', 'category']).columns)\nnumeric_cols_list = list(test.select_dtypes(include = ['float16', 'float64', 'int64']).columns)\n\n#categorical_cols_list = list(categorical_cols)\nfor col in categorical_cols_list: \n    if test[col].isnull().sum() == 0:\n        categorical_cols_list.remove(col)\n        \n\n#numeric_cols_list = list(numeric_cols)\nfor col in numeric_cols_list: \n    if test[col].isnull().sum() == 0:\n        numeric_cols_list.remove(col)\n\n#test.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_categorical = SimpleImputer(missing_values = np.nan, strategy = 'most_frequent')\n\ntest[categorical_cols_list]=imp_categorical.fit_transform(test[categorical_cols_list])\n\ntest[categorical_cols_list].isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace null values with zeros\ntest[numeric_cols_list[:88]] = test[numeric_cols_list[:88]].replace(np.nan, 0)\ntest[numeric_cols_list[88:]] = test[numeric_cols_list[88:]].replace(np.nan, 0)\n\n#replace with modes instead\n\n#test.fillna(modes, inplace = True)\n\ntest.isna().sum().sum()\n\nprint(f'Nulls in numerical: {test.isna().sum().sum()}')\n\ngc.collect() ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#categorical_cols = test.drop('customer_ID', axis =1).select_dtypes(include = ['object', 'category']).columns\n\ncategorical_cols = list(test.select_dtypes(include = ['object', 'category']).columns)\ncategorical_cols.remove('customer_ID')\n\nfor col in test[categorical_cols]:\n    test[col] = test[col].astype('category')\n\nfor col in test[categorical_cols]:\n    test[col] = test[col].cat.codes\n    \nprint('Done with categoricals')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_cols = test.select_dtypes(include = ['float16', 'float64', 'int64']).columns\n\nscaler = StandardScaler()\n\n# scaler.fit_transform(test[numeric_cols])\n# this doesn't work: not enough memory. use standard scaler on each feature, \n# one at a time\n\n#for col in test[numeric_cols]:\n    #test[col] = scaler.fit_transform(test[col].values.reshape(-1, 1))\n# this doesn't work because of memory constraints. divide the numerical columns in two\n# sets and scale each set one at a time:\n    \nfor col in test[numeric_cols[:88]]:\n    test[col] = scaler.fit_transform(test[col].values.reshape(-1, 1))\n    \nfor col in test[numeric_cols[88:]]:\n    test[col] = scaler.fit_transform(test[col].values.reshape(-1, 1))\n\nprint(len(test))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#x_test = test.drop('customer_ID', axis =1)\n\nx_test = test.groupby(['customer_ID']).nth(-1).reset_index(drop=True)\n\nprint('Test set defined successfully')\n\n# delete the test dataframe to free memory\n\ndel test\ngc.collect()\ntest = pd.DataFrame()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nxgb_model = joblib.load('../xgb model.pkl')\nxgb_model2 = joblib.load('../xgb model2.pkl')\nxgb_model3 = joblib.load('../xgb model3.pkl')\n\nxgb_predictions = xgb_model.predict_proba(x_test)[:,1]\nxgb_predictions2 = xgb_model2.predict_proba(x_test)[:,1]\nxgb_predictions3 = xgb_model3.predict_proba(x_test)[:,1]\n\ndel xgb_model\nxgb_model = None\ndel xgb_model2\nxgb_model2 = None\ndel xgb_model3\nxgb_model3 = None\ngc.collect()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample =  pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\n\nxgb_output = pd.DataFrame({'customer_ID': sample['customer_ID'], 'prediction': xgb_predictions}).to_csv('xgb_submission.csv', index=False)\nxgb_output2 = pd.DataFrame({'customer_ID': sample['customer_ID'], 'prediction': xgb_predictions2}).to_csv('xgb_submission2.csv', index=False)\nxgb_output3 = pd.DataFrame({'customer_ID': sample['customer_ID'], 'prediction': xgb_predictions3}).to_csv('xgb_submission3.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#average of predictions:\navg_predictions = pd.DataFrame({'customer_ID': sample['customer_ID'], 'prediction': np.round((xgb_predictions + xgb_predictions2 + xgb_predictions3)/3,2)}).to_csv('avg_submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#done","metadata":{},"execution_count":null,"outputs":[]}]}