{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3739819,"sourceType":"datasetVersion","datasetId":2231132}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport scipy\nfrom catboost import CatBoostClassifier\nfrom sklearn.preprocessing import LabelEncoder\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-13T15:31:05.646946Z","iopub.execute_input":"2024-02-13T15:31:05.647379Z","iopub.status.idle":"2024-02-13T15:31:08.146085Z","shell.execute_reply.started":"2024-02-13T15:31:05.647345Z","shell.execute_reply":"2024-02-13T15:31:08.144602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet(\"/kaggle/input/amex-data-integer-dtypes-parquet-format/train.parquet\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:31:08.148342Z","iopub.execute_input":"2024-02-13T15:31:08.149327Z","iopub.status.idle":"2024-02-13T15:31:27.506798Z","shell.execute_reply.started":"2024-02-13T15:31:08.149283Z","shell.execute_reply":"2024-02-13T15:31:27.506070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\ntrain = train.merge(train_label,how=\"inner\",on=\"customer_ID\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:31:27.508092Z","iopub.execute_input":"2024-02-13T15:31:27.508608Z","iopub.status.idle":"2024-02-13T15:31:30.899293Z","shell.execute_reply.started":"2024-02-13T15:31:27.508579Z","shell.execute_reply":"2024-02-13T15:31:30.898209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_parquet(\"/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:31:30.901886Z","iopub.execute_input":"2024-02-13T15:31:30.902208Z","iopub.status.idle":"2024-02-13T15:32:27.792189Z","shell.execute_reply.started":"2024-02-13T15:31:30.902181Z","shell.execute_reply":"2024-02-13T15:32:27.789813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in test.columns:\n    if test[col].dtype=='float16':\n        train[col] = train[col].astype('float32').round(decimals=2).astype('float16')\n        test[col] = test[col].astype('float32').round(decimals=2).astype('float16')","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:32:27.795142Z","iopub.execute_input":"2024-02-13T15:32:27.795516Z","iopub.status.idle":"2024-02-13T15:32:27.814818Z","shell.execute_reply.started":"2024-02-13T15:32:27.795487Z","shell.execute_reply":"2024-02-13T15:32:27.813551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train.drop(['customer_ID', 'S_2', 'target'], axis=1).columns.to_list()\ncat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnum_features = [col for col in features if col not in cat_features]","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:32:27.816335Z","iopub.execute_input":"2024-02-13T15:32:27.816931Z","iopub.status.idle":"2024-02-13T15:32:31.080040Z","shell.execute_reply.started":"2024-02-13T15:32:27.816896Z","shell.execute_reply":"2024-02-13T15:32:31.078945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_num_agg = train.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\ntrain_num_agg.columns = ['_'.join(x) for x in train_num_agg.columns]\ntrain_cat_agg = train.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\ntrain_cat_agg.columns = ['_'.join(x) for x in train_cat_agg.columns]\ntrain_target = (train.groupby(\"customer_ID\").tail(1).set_index('customer_ID', drop=True).sort_index()[\"target\"])\ntrain_agg = pd.concat([train_num_agg, train_cat_agg, train_target], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:32:31.081424Z","iopub.execute_input":"2024-02-13T15:32:31.081924Z","iopub.status.idle":"2024-02-13T15:33:42.123620Z","shell.execute_reply.started":"2024-02-13T15:32:31.081879Z","shell.execute_reply":"2024-02-13T15:33:42.122487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_num_agg = test.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\ntest_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\ntest_cat_agg = test.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\ntest_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\ntest_agg = pd.concat([test_num_agg, test_cat_agg], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:33:42.125130Z","iopub.execute_input":"2024-02-13T15:33:42.125457Z","iopub.status.idle":"2024-02-13T15:36:04.303115Z","shell.execute_reply.started":"2024-02-13T15:33:42.125427Z","shell.execute_reply":"2024-02-13T15:36:04.301927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_columns = ['D_66','D_42','D_49','D_73','D_76','R_9','B_29','D_87','D_88','D_106','R_26','D_108',\n               'D_53', 'D_110','D_111','B_39','B_42','D_132','D_134','D_135','D_136','D_137','D_138','D_142']","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:04.304646Z","iopub.execute_input":"2024-02-13T15:36:04.305438Z","iopub.status.idle":"2024-02-13T15:36:04.312492Z","shell.execute_reply.started":"2024-02-13T15:36:04.305394Z","shell.execute_reply":"2024-02-13T15:36:04.311075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_agg = train_agg.drop(columns=train_agg.columns[train_agg.columns.str.startswith(tuple(nan_columns))])\n\ntrain_agg.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:04.316466Z","iopub.execute_input":"2024-02-13T15:36:04.316816Z","iopub.status.idle":"2024-02-13T15:36:06.742706Z","shell.execute_reply.started":"2024-02-13T15:36:04.316787Z","shell.execute_reply":"2024-02-13T15:36:06.741618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\ncat_features = ['B_30_last', 'B_38_last', 'D_114_last', 'D_116_last', 'D_117_last',\n                'D_120_last', 'D_126_last', 'D_63_last', 'D_64_last', 'D_68_last']\nfor cat in cat_features:\n    train_agg[cat] = le.fit_transform(train_agg[cat])\n    test_agg[cat] = le.transform(test_agg[cat])","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:06.743873Z","iopub.execute_input":"2024-02-13T15:36:06.744175Z","iopub.status.idle":"2024-02-13T15:36:07.291353Z","shell.execute_reply.started":"2024-02-13T15:36:06.744148Z","shell.execute_reply":"2024-02-13T15:36:07.290309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.DataFrame(train_agg[\"target\"])\ntrain_agg = train_agg.drop(\"target\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:07.292545Z","iopub.execute_input":"2024-02-13T15:36:07.293640Z","iopub.status.idle":"2024-02-13T15:36:07.822814Z","shell.execute_reply.started":"2024-02-13T15:36:07.293604Z","shell.execute_reply":"2024-02-13T15:36:07.821634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_agg = test_agg[train_agg.columns.to_list()]\ntest_agg.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:07.824194Z","iopub.execute_input":"2024-02-13T15:36:07.824526Z","iopub.status.idle":"2024-02-13T15:36:11.704150Z","shell.execute_reply.started":"2024-02-13T15:36:07.824495Z","shell.execute_reply":"2024-02-13T15:36:11.702915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cbc = CatBoostClassifier(\n#     iterations= 5000,\n#     cat_features=cat_features,\n#     verbose=500,\n#     learning_rate=0.032251,\n#     depth=6,\n#     subsample = 0.800000011920929, \n#     min_data_in_leaf = 1,\n#     random_state=26\n# )\n# cbc.fit(train_agg, target)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:11.706858Z","iopub.execute_input":"2024-02-13T15:36:11.707673Z","iopub.status.idle":"2024-02-13T15:36:11.712412Z","shell.execute_reply.started":"2024-02-13T15:36:11.707628Z","shell.execute_reply":"2024-02-13T15:36:11.711617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_agg = train_agg.ffill().bfill()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:11.713972Z","iopub.execute_input":"2024-02-13T15:36:11.714738Z","iopub.status.idle":"2024-02-13T15:36:13.748789Z","shell.execute_reply.started":"2024-02-13T15:36:11.714670Z","shell.execute_reply":"2024-02-13T15:36:13.747753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_agg.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:13.750047Z","iopub.execute_input":"2024-02-13T15:36:13.750425Z","iopub.status.idle":"2024-02-13T15:36:13.775504Z","shell.execute_reply.started":"2024-02-13T15:36:13.750388Z","shell.execute_reply":"2024-02-13T15:36:13.774366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from imblearn.over_sampling import SMOTE\n# from sklearn.model_selection import train_test_split\n# X_train, X_test, y_train, y_test = train_test_split(train_agg, target, test_size=0.1, random_state=42, stratify=target)\n\n# smote = SMOTE(random_state=42)\n# X_train_resampled, y_train_resampled = smote.fit_resample(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:13.776850Z","iopub.execute_input":"2024-02-13T15:36:13.777211Z","iopub.status.idle":"2024-02-13T15:36:13.784164Z","shell.execute_reply.started":"2024-02-13T15:36:13.777172Z","shell.execute_reply":"2024-02-13T15:36:13.783166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:36:13.786012Z","iopub.execute_input":"2024-02-13T15:36:13.786762Z","iopub.status.idle":"2024-02-13T15:36:13.798563Z","shell.execute_reply.started":"2024-02-13T15:36:13.786721Z","shell.execute_reply":"2024-02-13T15:36:13.797661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx_0 = np.where(target == 0)[0]\nidx_1 = np.where(target == 1)[0]\n\n# Determine the number of examples to remove to achieve your desired ratio\n# For example, to achieve a 1:2 ratio of 0s to 1s\ndesired_ratio = 1.3  # This means 1 zero for every 2 ones\nn_zeros = len(idx_0)\nn_ones_needed = n_zeros * desired_ratio\n\nif len(idx_1) > n_ones_needed:\n    # Randomly select indices of ones to keep\n    idx_1_to_keep = np.random.choice(idx_1, size=n_ones_needed, replace=False)\n    new_indices = np.concatenate([idx_0, idx_1_to_keep])\nelse:\n    new_indices = np.concatenate([idx_0, idx_1])\n\n# Create the undersampled dataset\nX_train_us = train_agg.iloc[new_indices]\ny_train_us = target.iloc[new_indices]","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:38:22.405811Z","iopub.execute_input":"2024-02-13T15:38:22.406983Z","iopub.status.idle":"2024-02-13T15:38:23.488608Z","shell.execute_reply.started":"2024-02-13T15:38:22.406930Z","shell.execute_reply":"2024-02-13T15:38:23.487585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cbc = CatBoostClassifier(iterations = 5000, random_state = 26, verbose = 500)\ncbc.fit(X_train_us, y_train_us)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:39:01.455599Z","iopub.execute_input":"2024-02-13T15:39:01.456044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cbc_predict = cbc.predict_proba(test_agg)\npred = cbc_predict[:,1]","metadata":{"execution":{"iopub.status.busy":"2024-02-13T15:21:58.937606Z","iopub.status.idle":"2024-02-13T15:21:58.938645Z","shell.execute_reply.started":"2024-02-13T15:21:58.938385Z","shell.execute_reply":"2024-02-13T15:21:58.938415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['prediction'] = pred\nsub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}