{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport regex as re\nimport gc\nimport matplotlib.pyplot as plt\nimport pyarrow.feather as feather\nimport seaborn as sns\nimport datatable as dt\nimport random\nimport pyarrow.feather as feather\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score  \nfrom sklearn.metrics import precision_score                         \nfrom sklearn.metrics import recall_score\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-19T16:48:04.769069Z","iopub.execute_input":"2022-08-19T16:48:04.769779Z","iopub.status.idle":"2022-08-19T16:48:06.158147Z","shell.execute_reply.started":"2022-08-19T16:48:04.769693Z","shell.execute_reply":"2022-08-19T16:48:06.157114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data preparation","metadata":{}},{"cell_type":"markdown","source":"Thanks Taddar for parquet dataset from [here](https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format)\n\nThanks huseyincotel for agg data pickle dataset from [here](https://www.kaggle.com/datasets/huseyincot/amex-agg-data-pickle)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T05:48:36.731099Z","iopub.execute_input":"2022-08-14T05:48:36.732217Z","iopub.status.idle":"2022-08-14T05:48:36.737882Z","shell.execute_reply.started":"2022-08-14T05:48:36.732180Z","shell.execute_reply":"2022-08-14T05:48:36.736720Z"}}},{"cell_type":"code","source":"#preData  = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows = 10000, low_memory=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:48:06.160196Z","iopub.execute_input":"2022-08-19T16:48:06.160540Z","iopub.status.idle":"2022-08-19T16:48:06.166025Z","shell.execute_reply.started":"2022-08-19T16:48:06.160502Z","shell.execute_reply":"2022-08-19T16:48:06.164869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preData = pd.read_pickle('/kaggle/input/amex-agg-data-pickle/train_agg.pkl',compression='gzip')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:48:06.167496Z","iopub.execute_input":"2022-08-19T16:48:06.168679Z","iopub.status.idle":"2022-08-19T16:48:24.543596Z","shell.execute_reply.started":"2022-08-19T16:48:06.168643Z","shell.execute_reply":"2022-08-19T16:48:24.542642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data(data):\n    # Data cleaning: remove >75% na\n    cols_with_75pc_missing = [col for col in data.columns if data[col].isna().sum() >  0.75*len(data.index)]\n    data = data.drop(columns=cols_with_75pc_missing)\n    drop_cols = ['B_29','S_9','D_83','D_69','B_39','S_13']\n    for i in drop_cols:\n        if i in data.columns:\n            data = data.drop(columns=i)\n            \n    # Drop unused column to reduce memory   \n    if \"target\" in data.columns:\n        data = data.drop(columns=[\"target\"])\n    if \"customer_ID\" in data.columns:\n        data = data.drop(columns=[\"customer_ID\"])\n    if \"S_2\" in data.columns:\n        data = data.drop(columns=[\"S_2\"])\n\n    # Remove outliers which is greater that 2.0 for float values\n    for i in data.columns:\n        if data[i].dtype == \"float16\" or data[i].dtype == \"float64\":\n            m = data[i].mean()\n            if np.isnan(m):\n                m = data[i].median()\n            data[i] = data[i].apply(lambda x: x if x < 2.0 else m)\n        \n    # Data imputation: Fill up NaN\n    for i in data.columns:\n        if data[i].isna().sum() > 0:\n            if i == \"D_63_last\":\n                data[i] = data[i].fillna(\"CO\")\n            elif i == \"D_64_last\":\n                data[i] = data[i].fillna(\"O\")\n            elif data[i].dtype == \"object\":\n                data[i] = data[i].fillna(\"-1\")\n            else:\n                data[i] = data[i].fillna(data[i].mean())\n    \n    # Convert categorical into numerical\n    le = LabelEncoder()\n    data['D_63_last'] = le.fit_transform(data['D_63_last'])\n    data['D_64_last'] = le.fit_transform(data['D_64_last'])\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:48:24.546268Z","iopub.execute_input":"2022-08-19T16:48:24.546597Z","iopub.status.idle":"2022-08-19T16:48:24.560724Z","shell.execute_reply.started":"2022-08-19T16:48:24.546561Z","shell.execute_reply":"2022-08-19T16:48:24.558518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preData = process_data(preData)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:48:24.561948Z","iopub.execute_input":"2022-08-19T16:48:24.563174Z","iopub.status.idle":"2022-08-19T16:50:53.010313Z","shell.execute_reply.started":"2022-08-19T16:48:24.563136Z","shell.execute_reply":"2022-08-19T16:50:53.009351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs('/tmp',exist_ok=True)\nfeather.write_feather(preData, '/tmp/preData.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:50:53.011880Z","iopub.execute_input":"2022-08-19T16:50:53.012260Z","iopub.status.idle":"2022-08-19T16:51:01.603183Z","shell.execute_reply.started":"2022-08-19T16:50:53.012225Z","shell.execute_reply":"2022-08-19T16:51:01.601627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preData = feather.read_feather('/tmp/preData.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:51:01.609022Z","iopub.execute_input":"2022-08-19T16:51:01.611982Z","iopub.status.idle":"2022-08-19T16:51:07.169689Z","shell.execute_reply.started":"2022-08-19T16:51:01.611911Z","shell.execute_reply":"2022-08-19T16:51:07.168708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{}},{"cell_type":"code","source":"# Train test split\npreLabel  = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', low_memory=True)\nX_train, X_val, y_train, y_val = train_test_split(preData,preLabel['target'],test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:51:07.171253Z","iopub.execute_input":"2022-08-19T16:51:07.171611Z","iopub.status.idle":"2022-08-19T16:51:11.613070Z","shell.execute_reply.started":"2022-08-19T16:51:07.171560Z","shell.execute_reply":"2022-08-19T16:51:11.612100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del preData, preLabel\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:51:11.614571Z","iopub.execute_input":"2022-08-19T16:51:11.614939Z","iopub.status.idle":"2022-08-19T16:51:11.749018Z","shell.execute_reply.started":"2022-08-19T16:51:11.614885Z","shell.execute_reply":"2022-08-19T16:51:11.747909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:51:11.753309Z","iopub.execute_input":"2022-08-19T16:51:11.753586Z","iopub.status.idle":"2022-08-19T16:51:13.353116Z","shell.execute_reply.started":"2022-08-19T16:51:11.753561Z","shell.execute_reply":"2022-08-19T16:51:13.352068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CatBoost","metadata":{}},{"cell_type":"code","source":"# # iter = 100, 18 seconds, 3000+GPU = 2 mins\nimport catboost as cat\nfrom catboost import CatBoostClassifier\nfrom catboost import CatBoostRegressor\n\nclf = CatBoostClassifier(\n    iterations=5000,\n    metric_period = 100,\n    task_type = 'GPU',\n    eval_metric = 'NormalizedGini', \n    bagging_temperature = 0.2\n)\nclf.fit(X_train,y_train, use_best_model=True, eval_set=(X_val,y_val),verbose=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:56:34.914079Z","iopub.execute_input":"2022-08-19T16:56:34.914444Z","iopub.status.idle":"2022-08-19T17:00:51.869400Z","shell.execute_reply.started":"2022-08-19T16:56:34.914413Z","shell.execute_reply":"2022-08-19T17:00:51.868590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.save_model('catboost')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:00:51.873331Z","iopub.execute_input":"2022-08-19T17:00:51.875599Z","iopub.status.idle":"2022-08-19T17:00:51.919870Z","shell.execute_reply.started":"2022-08-19T17:00:51.875561Z","shell.execute_reply":"2022-08-19T17:00:51.918983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = clf.predict_proba(X_val)\nrounded_predictions = np.argmax(prediction, axis=-1)\nc_matrix = confusion_matrix(y_val,rounded_predictions)\ndt_acc = c_matrix.trace()/c_matrix.sum()\nprint(c_matrix)\nprint(dt_acc)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:00:51.920954Z","iopub.execute_input":"2022-08-19T17:00:51.921321Z","iopub.status.idle":"2022-08-19T17:00:53.341781Z","shell.execute_reply.started":"2022-08-19T17:00:51.921285Z","shell.execute_reply":"2022-08-19T17:00:53.340609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rounded_predictions = np.argmax(prediction, axis=-1)\nnp.unique(rounded_predictions, return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:00:53.344965Z","iopub.execute_input":"2022-08-19T17:00:53.346002Z","iopub.status.idle":"2022-08-19T17:00:53.357178Z","shell.execute_reply.started":"2022-08-19T17:00:53.345964Z","shell.execute_reply":"2022-08-19T17:00:53.356029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(accuracy_score(y_val,rounded_predictions))\nprint(precision_score(y_val,rounded_predictions))\nprint(recall_score(y_val,rounded_predictions))","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:00:53.358998Z","iopub.execute_input":"2022-08-19T17:00:53.359371Z","iopub.status.idle":"2022-08-19T17:00:53.442907Z","shell.execute_reply.started":"2022-08-19T17:00:53.359335Z","shell.execute_reply":"2022-08-19T17:00:53.441764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#cleanup\ndel X_train, X_val, y_train, y_val\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:00:53.444329Z","iopub.execute_input":"2022-08-19T17:00:53.444892Z","iopub.status.idle":"2022-08-19T17:00:53.564239Z","shell.execute_reply.started":"2022-08-19T17:00:53.444855Z","shell.execute_reply":"2022-08-19T17:00:53.562979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:00:53.565581Z","iopub.execute_input":"2022-08-19T17:00:53.566361Z","iopub.status.idle":"2022-08-19T17:00:53.679055Z","shell.execute_reply.started":"2022-08-19T17:00:53.566320Z","shell.execute_reply":"2022-08-19T17:00:53.677968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Final validation","metadata":{}},{"cell_type":"code","source":"#testData = pd.read_parquet('/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet')\n#testCustomer  = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv', usecols=['customer_ID'], low_memory=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:56:19.816698Z","iopub.status.idle":"2022-08-19T16:56:19.817840Z","shell.execute_reply.started":"2022-08-19T16:56:19.817591Z","shell.execute_reply":"2022-08-19T16:56:19.817615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testData = pd.read_pickle('/kaggle/input/amex-agg-data-pickle/test_agg.pkl',compression='gzip')\ntestCustomer  = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv', usecols=['customer_ID'], low_memory=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:00:53.680802Z","iopub.execute_input":"2022-08-19T17:00:53.681224Z","iopub.status.idle":"2022-08-19T17:01:29.829573Z","shell.execute_reply.started":"2022-08-19T17:00:53.681165Z","shell.execute_reply":"2022-08-19T17:01:29.828584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#feather.write_feather(testData, '/tmp/testData.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T16:56:19.821499Z","iopub.status.idle":"2022-08-19T16:56:19.822469Z","shell.execute_reply.started":"2022-08-19T16:56:19.822207Z","shell.execute_reply":"2022-08-19T16:56:19.822232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data preparation","metadata":{}},{"cell_type":"code","source":"testData = process_data(testData)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:01:29.831146Z","iopub.execute_input":"2022-08-19T17:01:29.831500Z","iopub.status.idle":"2022-08-19T17:06:45.508563Z","shell.execute_reply.started":"2022-08-19T17:01:29.831462Z","shell.execute_reply":"2022-08-19T17:06:45.507542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feather.write_feather(testData, '/tmp/processedTestData.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:06:45.512031Z","iopub.execute_input":"2022-08-19T17:06:45.512442Z","iopub.status.idle":"2022-08-19T17:07:04.791224Z","shell.execute_reply.started":"2022-08-19T17:06:45.512404Z","shell.execute_reply":"2022-08-19T17:07:04.790199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:07:04.792797Z","iopub.execute_input":"2022-08-19T17:07:04.793159Z","iopub.status.idle":"2022-08-19T17:07:04.919437Z","shell.execute_reply.started":"2022-08-19T17:07:04.793123Z","shell.execute_reply":"2022-08-19T17:07:04.918390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Final Prediction","metadata":{}},{"cell_type":"code","source":"prediction = clf.predict_proba(testData)\nfinal_predictions = prediction[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:07:04.922864Z","iopub.execute_input":"2022-08-19T17:07:04.923205Z","iopub.status.idle":"2022-08-19T17:07:17.370752Z","shell.execute_reply.started":"2022-08-19T17:07:04.923179Z","shell.execute_reply":"2022-08-19T17:07:17.369708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'customer_ID': testCustomer.customer_ID, 'prediction': final_predictions})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:07:17.372816Z","iopub.execute_input":"2022-08-19T17:07:17.373584Z","iopub.status.idle":"2022-08-19T17:07:20.212297Z","shell.execute_reply.started":"2022-08-19T17:07:17.373541Z","shell.execute_reply":"2022-08-19T17:07:20.211195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"value, count = np.unique(final_predictions,return_counts=True)\nprint(\"Final predictions\",final_predictions, final_predictions.mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-19T17:07:20.213892Z","iopub.execute_input":"2022-08-19T17:07:20.214298Z","iopub.status.idle":"2022-08-19T17:07:20.322631Z","shell.execute_reply.started":"2022-08-19T17:07:20.214263Z","shell.execute_reply":"2022-08-19T17:07:20.321599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}