{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport gc\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:52:47.388508Z","iopub.execute_input":"2022-12-12T13:52:47.388962Z","iopub.status.idle":"2022-12-12T13:52:47.400485Z","shell.execute_reply.started":"2022-12-12T13:52:47.388927Z","shell.execute_reply":"2022-12-12T13:52:47.399089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing data","metadata":{}},{"cell_type":"code","source":"# Import dataset\ntrain_dataset = pd.read_feather('../input/amexfeather/train_data.ftr')\n\n# Keep the latest statement features for each customer\ntrain_dataset = train_dataset.groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:52:47.403278Z","iopub.execute_input":"2022-12-12T13:52:47.403748Z","iopub.status.idle":"2022-12-12T13:52:59.139688Z","shell.execute_reply.started":"2022-12-12T13:52:47.403702Z","shell.execute_reply":"2022-12-12T13:52:59.138543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization ","metadata":{}},{"cell_type":"code","source":"# First 5 rows of the dataset\ntrain_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:52:59.141393Z","iopub.execute_input":"2022-12-12T13:52:59.141745Z","iopub.status.idle":"2022-12-12T13:52:59.171360Z","shell.execute_reply.started":"2022-12-12T13:52:59.141715Z","shell.execute_reply":"2022-12-12T13:52:59.169868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get train set details\ntrain_dataset.info(max_cols=191,show_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:52:59.173966Z","iopub.execute_input":"2022-12-12T13:52:59.174403Z","iopub.status.idle":"2022-12-12T13:52:59.673341Z","shell.execute_reply.started":"2022-12-12T13:52:59.174368Z","shell.execute_reply":"2022-12-12T13:52:59.672214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Describe train dataset\ntrain_dataset.describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:52:59.675266Z","iopub.execute_input":"2022-12-12T13:52:59.677728Z","iopub.status.idle":"2022-12-12T13:53:15.057396Z","shell.execute_reply.started":"2022-12-12T13:52:59.677676Z","shell.execute_reply":"2022-12-12T13:53:15.056077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore categorical and non categorical features\n\n* D_* = Delinquency variables (bad or criminal behaviour, especially among young people)\n* S_* = Spend variables\n* P_* = Payment variables\n* B_* = Balance variables\n* R_* = Risk variables\n\n### categorical\n\n['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{}},{"cell_type":"code","source":"categorical_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nnum_cols = [col for col in train_dataset.columns if col not in categorical_cols + [\"target\"]]\n\nprint(f'Total number of features: {1}')\nprint(f'Total number of categorical features: {len(categorical_cols)}')\nprint(f'Total number of continuos features: {len(num_cols)}')","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:53:15.058699Z","iopub.execute_input":"2022-12-12T13:53:15.059021Z","iopub.status.idle":"2022-12-12T13:53:15.067892Z","shell.execute_reply.started":"2022-12-12T13:53:15.058993Z","shell.execute_reply":"2022-12-12T13:53:15.066427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the target variable\nsns.countplot(x = 'target', data = train_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:53:15.069315Z","iopub.execute_input":"2022-12-12T13:53:15.069660Z","iopub.status.idle":"2022-12-12T13:53:19.288781Z","shell.execute_reply.started":"2022-12-12T13:53:15.069627Z","shell.execute_reply":"2022-12-12T13:53:19.287498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize categorial features\nplt.figure(figsize=(20, 30))\nfor i, k in enumerate(categorical_cols):\n    plt.subplot(6, 2, i+1)\n    temp_val = pd.DataFrame(train_dataset[k].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp_val.index.name = 'value'\n    temp_val.reset_index(inplace=True)\n    plt.bar(temp_val.index, temp_val['count'], alpha=0.5)\n    plt.xlabel(k)\n    plt.ylabel('frequency')\n    plt.xticks(temp_val.index, temp_val.value)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:53:19.290646Z","iopub.execute_input":"2022-12-12T13:53:19.291147Z","iopub.status.idle":"2022-12-12T13:53:20.826129Z","shell.execute_reply.started":"2022-12-12T13:53:19.291110Z","shell.execute_reply":"2022-12-12T13:53:20.825137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize categorial features with targets\nplt.figure(figsize=(20, 30))\nfor i, f in enumerate(categorical_cols):\n    plt.subplot(6, 2, i+1)\n    temp = pd.DataFrame(train_dataset[f][train_dataset.target == 0].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp.index.name = 'value'\n    temp.reset_index(inplace=True)\n    plt.bar(temp.index, temp['count'], alpha=0.5, label='target=0')\n    temp = pd.DataFrame(train_dataset[f][train_dataset.target == 1].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp.index.name = 'value'\n    temp.reset_index(inplace=True)\n    plt.bar(temp.index, temp['count'], alpha=0.5, label='target=1')\n    plt.xlabel(f)\n    plt.ylabel('frequency')\n    plt.legend()\n    plt.xticks(temp.index, temp.value)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:53:20.827580Z","iopub.execute_input":"2022-12-12T13:53:20.828125Z","iopub.status.idle":"2022-12-12T13:53:23.315678Z","shell.execute_reply.started":"2022-12-12T13:53:20.828089Z","shell.execute_reply":"2022-12-12T13:53:23.314457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize non-categorial features\nfor i, l in enumerate(num_cols):\n    if i % 4 == 0: \n        if i > 0: plt.show()\n        plt.figure(figsize=(20, 3))\n    plt.subplot(1, 4, i % 4 + 1)\n    plt.hist(train_dataset[l], bins=200, color='#008000')\n    plt.xlabel(l)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:53:23.319735Z","iopub.execute_input":"2022-12-12T13:53:23.320627Z","iopub.status.idle":"2022-12-12T13:54:57.924523Z","shell.execute_reply.started":"2022-12-12T13:53:23.320577Z","shell.execute_reply":"2022-12-12T13:54:57.923250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Aggregated profile features\nDelinquency = [d for d in train_dataset.columns if d.startswith('D_')]\nSpend = [s for s in train_dataset.columns if s.startswith('S_')]\nPayment = [p for p in train_dataset.columns if p.startswith('P_')]\nBalance = [b for b in train_dataset.columns if b.startswith('B_')]\nRisk = [r for r in train_dataset.columns if r.startswith('R_')]\nDict = {'Delinquency': len(Delinquency), 'Spend': len(Spend), 'Payment': len(Payment), 'Balance': len(Balance), 'Risk': len(Risk),}\n\nplt.figure(figsize=(10,5))\nsns.barplot(x=list(Dict.keys()), y=list(Dict.values()));","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:54:57.925755Z","iopub.execute_input":"2022-12-12T13:54:57.926081Z","iopub.status.idle":"2022-12-12T13:54:58.146457Z","shell.execute_reply.started":"2022-12-12T13:54:57.926053Z","shell.execute_reply":"2022-12-12T13:54:58.145224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NaN_Val = np.array(train_dataset.isnull().sum())\nNaN_prec = np.array((train_dataset.isnull().sum() * 100 / len(train_dataset)).round(2))\nNaN_Col = pd.DataFrame([np.array(list(train_dataset.columns)).T,NaN_Val.T,NaN_prec.T,np.array(list(train_dataset.dtypes)).T], index=['Features','Num of Missing values','Percentage','DataType']\n).transpose()\npd.set_option('display.max_rows', None)\nNaN_Col","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:54:58.148348Z","iopub.execute_input":"2022-12-12T13:54:58.148813Z","iopub.status.idle":"2022-12-12T13:54:59.104724Z","shell.execute_reply.started":"2022-12-12T13:54:58.148769Z","shell.execute_reply":"2022-12-12T13:54:59.101261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop variables with missing values >=70% in the train dataframe\ni=0\nfor col in train_dataset.columns:\n    if (train_dataset[col].isnull().sum()/len(train_dataset[col])*100) >=70:\n        print(\"Dropping column\", col)\n        train_dataset.drop(labels=col,axis=1,inplace=True)\n        i=i+1\n        \nprint(\"Total number of columns dropped in train dataframe\", i)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:54:59.106336Z","iopub.execute_input":"2022-12-12T13:54:59.107116Z","iopub.status.idle":"2022-12-12T13:55:07.069728Z","shell.execute_reply.started":"2022-12-12T13:54:59.107066Z","shell.execute_reply":"2022-12-12T13:55:07.068253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset[\"S_2\"] = train_dataset[\"S_2\"].astype('datetime64[ns]')\ntrain_dataset[\"Day of week\"] = train_dataset[\"S_2\"].dt.dayofweek\ntrain_dataset[\"Year\"] = train_dataset[\"S_2\"].dt.year\ntrain_dataset[\"Month\"] = train_dataset[\"S_2\"].dt.month\ntrain_dataset[\"Day\"] = train_dataset[\"S_2\"].dt.day","metadata":{"execution":{"iopub.status.busy":"2022-12-12T13:58:59.754036Z","iopub.execute_input":"2022-12-12T13:58:59.754568Z","iopub.status.idle":"2022-12-12T13:58:59.941298Z","shell.execute_reply.started":"2022-12-12T13:58:59.754531Z","shell.execute_reply":"2022-12-12T13:58:59.939747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.drop(axis=1, columns=['S_2'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:04:57.792337Z","iopub.execute_input":"2022-12-12T14:04:57.792882Z","iopub.status.idle":"2022-12-12T14:04:58.260071Z","shell.execute_reply.started":"2022-12-12T14:04:57.792827Z","shell.execute_reply":"2022-12-12T14:04:58.258576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_col = np.array(['P_2','S_3','B_2','D_41','D_43','B_3','D_44','D_45','D_46','D_48','D_50','S_7','D_56','S_9','B_6','B_8','D_52','P_3','D_54','D_55','B_13','D_59','D_61','B_15','D_62','B_16','B_17','D_77','B_19','B_20','D_69','B_22','D_70','D_72','D_74','R_7','B_25','B_26','D_78','D_79','D_80','B_27','D_81','R_12','D_105','S_27','D_83','R_14','D_84','D_86','R_20','B_33','D_89','D_91','S_22','S_23','S_24','S_25','S_26','D_102','D_103','D_104','D_107','B_37','R_27','D_109','D_112','B_40','D_113','D_115','D_118','D_119','D_121','D_122','D_123','D_124','D_125','D_128','D_129','B_41','D_130','D_131','D_133','D_139','D_140','D_141','D_143','D_144','D_145'])\n\nfor col in selected_col:\n    train_dataset[col] = train_dataset[col].fillna(train_dataset[col].median())","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:29:43.097691Z","iopub.execute_input":"2022-12-12T14:29:43.098092Z","iopub.status.idle":"2022-12-12T14:29:44.800882Z","shell.execute_reply.started":"2022-12-12T14:29:43.098060Z","shell.execute_reply":"2022-12-12T14:29:44.799682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selcted_col2 = np.array(['D_68','B_30','B_38','D_64','D_114','D_116','D_117','D_120','D_126'])\n\nfor col2 in selcted_col2:\n    train_dataset[col2] =  train_dataset[col2].fillna(train_dataset[col2].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:29:50.657375Z","iopub.execute_input":"2022-12-12T14:29:50.657755Z","iopub.status.idle":"2022-12-12T14:29:50.715724Z","shell.execute_reply.started":"2022-12-12T14:29:50.657726Z","shell.execute_reply":"2022-12-12T14:29:50.714324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check null values again\nprint(train_dataset.isnull().sum().to_string())","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:30:32.321566Z","iopub.execute_input":"2022-12-12T14:30:32.322015Z","iopub.status.idle":"2022-12-12T14:30:32.685684Z","shell.execute_reply.started":"2022-12-12T14:30:32.321984Z","shell.execute_reply":"2022-12-12T14:30:32.684266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:31:16.022546Z","iopub.execute_input":"2022-12-12T14:31:16.022961Z","iopub.status.idle":"2022-12-12T14:31:16.031619Z","shell.execute_reply.started":"2022-12-12T14:31:16.022927Z","shell.execute_reply":"2022-12-12T14:31:16.030404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:31:18.762703Z","iopub.execute_input":"2022-12-12T14:31:18.763235Z","iopub.status.idle":"2022-12-12T14:31:18.794453Z","shell.execute_reply.started":"2022-12-12T14:31:18.763195Z","shell.execute_reply":"2022-12-12T14:31:18.793020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Dataset","metadata":{}},{"cell_type":"code","source":"test_dataset = pd.read_feather('../input/amexfeather/test_data.ftr')\n# Keep the latest statement features for each customer\ntest_dataset = test_dataset.groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:32:55.543577Z","iopub.execute_input":"2022-12-12T14:32:55.544044Z","iopub.status.idle":"2022-12-12T14:33:38.559680Z","shell.execute_reply.started":"2022-12-12T14:32:55.544009Z","shell.execute_reply":"2022-12-12T14:33:38.558185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:40:40.434706Z","iopub.execute_input":"2022-12-12T14:40:40.435200Z","iopub.status.idle":"2022-12-12T14:40:40.466876Z","shell.execute_reply.started":"2022-12-12T14:40:40.435145Z","shell.execute_reply":"2022-12-12T14:40:40.465733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:40:44.144471Z","iopub.execute_input":"2022-12-12T14:40:44.144896Z","iopub.status.idle":"2022-12-12T14:40:44.152746Z","shell.execute_reply.started":"2022-12-12T14:40:44.144860Z","shell.execute_reply":"2022-12-12T14:40:44.151373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NaN_Val2 = np.array(test_dataset.isnull().sum())\nNaN_prec2 = np.array((test_dataset.isnull().sum() * 100 / len(test_dataset)).round(2))\nNaN_Col2 = pd.DataFrame([np.array(list(test_dataset.columns)).T,NaN_Val2.T,NaN_prec2.T,np.array(list(test_dataset.dtypes)).T], index=['Features','Num of Missing values','Percentage','DataType']\n).transpose()\npd.set_option('display.max_rows', None)\n\nNaN_Col2","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:40:46.342686Z","iopub.execute_input":"2022-12-12T14:40:46.343130Z","iopub.status.idle":"2022-12-12T14:40:48.272722Z","shell.execute_reply.started":"2022-12-12T14:40:46.343093Z","shell.execute_reply":"2022-12-12T14:40:48.271505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = test_dataset.drop(['D_42', 'D_49', 'D_53', 'D_66', 'D_73', 'D_76', 'R_9', 'D_82', 'B_29', 'D_87', 'D_88', 'D_106', 'R_26', 'D_108', 'D_110', 'D_111', 'B_39', 'B_42', 'D_132', 'D_134', 'D_135', 'D_136', 'D_137', 'D_138', 'D_142'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:55:36.281687Z","iopub.execute_input":"2022-12-12T14:55:36.282204Z","iopub.status.idle":"2022-12-12T14:55:37.195126Z","shell.execute_reply.started":"2022-12-12T14:55:36.282154Z","shell.execute_reply":"2022-12-12T14:55:37.193981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset[\"S_2\"] = test_dataset[\"S_2\"].astype('datetime64[ns]')\ntest_dataset[\"Day of week\"] = test_dataset[\"S_2\"].dt.dayofweek\ntest_dataset[\"Year\"] = test_dataset[\"S_2\"].dt.year\ntest_dataset[\"Month\"] = test_dataset[\"S_2\"].dt.month\ntest_dataset[\"Day\"] = test_dataset[\"S_2\"].dt.day","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:55:42.577425Z","iopub.execute_input":"2022-12-12T14:55:42.577850Z","iopub.status.idle":"2022-12-12T14:55:42.984210Z","shell.execute_reply.started":"2022-12-12T14:55:42.577819Z","shell.execute_reply":"2022-12-12T14:55:42.983094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.drop(axis=1, columns=['S_2'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T14:55:46.036484Z","iopub.execute_input":"2022-12-12T14:55:46.036884Z","iopub.status.idle":"2022-12-12T14:55:46.933009Z","shell.execute_reply.started":"2022-12-12T14:55:46.036852Z","shell.execute_reply":"2022-12-12T14:55:46.931733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_col = np.array(['P_2','S_3','B_2','S_12', 'S_17','D_41','D_43','B_3','D_44','D_45','D_46','D_48','D_50','S_7','D_56','S_9','B_6','B_8','D_52','P_3','D_54','D_55','B_13','D_59','D_61','B_15','D_62','B_16','B_17','D_77','B_19','B_20','D_69','B_22','D_70','D_72','D_74','R_7','B_25','B_26','D_78','D_79','D_80','B_27','D_81','R_12','D_105','S_27','D_83','R_14','D_84','D_86','R_20','B_33','D_89','D_91','S_22','S_23','S_24','S_25','S_26','D_102','D_103','D_104','D_107','B_37','R_27','D_109','D_112','B_40','D_113','D_115','D_118','D_119','D_121','D_122','D_123','D_124','D_125','D_128','D_129','B_41','D_130','D_131','D_133','D_139','D_140','D_141','D_143','D_144','D_145'])\n\nfor col in selected_col:\n    test_dataset[col] = test_dataset[col].fillna(test_dataset[col].median())","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:01:19.072548Z","iopub.execute_input":"2022-12-12T15:01:19.072995Z","iopub.status.idle":"2022-12-12T15:01:22.438823Z","shell.execute_reply.started":"2022-12-12T15:01:19.072962Z","shell.execute_reply":"2022-12-12T15:01:22.437451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selcted_col2 = np.array(['D_68','B_30','B_38','D_64','D_114','D_116','D_117','D_120','D_126'])\n\nfor col2 in selcted_col2:\n    test_dataset[col2] =  test_dataset[col2].fillna(test_dataset[col2].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:01:24.618115Z","iopub.execute_input":"2022-12-12T15:01:24.618520Z","iopub.status.idle":"2022-12-12T15:01:24.712483Z","shell.execute_reply.started":"2022-12-12T15:01:24.618490Z","shell.execute_reply":"2022-12-12T15:01:24.711150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check null values again\nprint(test_dataset.isnull().sum().to_string())","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:01:27.465969Z","iopub.execute_input":"2022-12-12T15:01:27.466429Z","iopub.status.idle":"2022-12-12T15:01:28.184056Z","shell.execute_reply.started":"2022-12-12T15:01:27.466390Z","shell.execute_reply":"2022-12-12T15:01:28.182661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:01:33.655472Z","iopub.execute_input":"2022-12-12T15:01:33.655875Z","iopub.status.idle":"2022-12-12T15:01:33.663565Z","shell.execute_reply.started":"2022-12-12T15:01:33.655844Z","shell.execute_reply":"2022-12-12T15:01:33.662558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:01:36.520262Z","iopub.execute_input":"2022-12-12T15:01:36.520670Z","iopub.status.idle":"2022-12-12T15:01:36.552864Z","shell.execute_reply.started":"2022-12-12T15:01:36.520625Z","shell.execute_reply":"2022-12-12T15:01:36.551468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\nenc = OrdinalEncoder()\ncategorical_cols.remove('D_66')\n\ntrain_dataset[categorical_cols] = enc.fit_transform(train_dataset[categorical_cols])\ntest_dataset[categorical_cols] = enc.transform(test_dataset[categorical_cols])","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:01:39.553262Z","iopub.execute_input":"2022-12-12T15:01:39.553675Z","iopub.status.idle":"2022-12-12T15:01:41.584882Z","shell.execute_reply.started":"2022-12-12T15:01:39.553643Z","shell.execute_reply":"2022-12-12T15:01:41.583649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset_without_target = train_dataset.drop([\"target\"],axis=1)\n\ncor_matrix = train_dataset_without_target.corr()\ncol_core = set()\n\nfor i in range(len(cor_matrix.columns)):\n    for j in range(i):\n        if(cor_matrix.iloc[i, j] > 0.9):\n            col_name = cor_matrix.columns[i]\n            col_core.add(col_name)\ncol_core","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:01:44.230424Z","iopub.execute_input":"2022-12-12T15:01:44.230958Z","iopub.status.idle":"2022-12-12T15:02:20.326698Z","shell.execute_reply.started":"2022-12-12T15:01:44.230915Z","shell.execute_reply":"2022-12-12T15:02:20.325243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = train_dataset.drop(col_core, axis=1)\ntest_dataset = test_dataset.drop(col_core, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:03:32.771604Z","iopub.execute_input":"2022-12-12T15:03:32.772105Z","iopub.status.idle":"2022-12-12T15:03:33.932008Z","shell.execute_reply.started":"2022-12-12T15:03:32.772050Z","shell.execute_reply":"2022-12-12T15:03:33.930839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:03:36.102666Z","iopub.execute_input":"2022-12-12T15:03:36.103100Z","iopub.status.idle":"2022-12-12T15:03:36.111435Z","shell.execute_reply.started":"2022-12-12T15:03:36.103056Z","shell.execute_reply":"2022-12-12T15:03:36.109972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_columns = [col for col in train_dataset.columns if col not in [\"target\"]]\n\nX = train_dataset[num_columns]\ny = train_dataset['target']\n\nprint(f\"X shape is = {X.shape}\" )\nprint(f\"Y shape is = {y.shape}\" )","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:03:38.935398Z","iopub.execute_input":"2022-12-12T15:03:38.935811Z","iopub.status.idle":"2022-12-12T15:03:39.318485Z","shell.execute_reply.started":"2022-12-12T15:03:38.935780Z","shell.execute_reply":"2022-12-12T15:03:39.316950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nx_train,x_test,y_train,y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\nprint(f\"X_train shape is = {x_train.shape}\" )\nprint(f\"Y_train shape is = {y_train.shape}\" )\nprint(f\"X_test shape is = {x_test.shape}\" )\nprint(f\"Y_test shape is = {y_test.shape}\" )","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:03:41.696259Z","iopub.execute_input":"2022-12-12T15:03:41.696733Z","iopub.status.idle":"2022-12-12T15:03:43.104460Z","shell.execute_reply.started":"2022-12-12T15:03:41.696695Z","shell.execute_reply":"2022-12-12T15:03:43.103233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n\nd_train = lgb.Dataset(x_train, label=y_train, categorical_feature = categorical_cols)\n\nparams = {'objective': 'binary','n_estimators': 1200,'metric': 'binary_logloss','boosting': 'gbdt','num_leaves': 90,'reg_lambda' : 50,'colsample_bytree': 0.19,'learning_rate': 0.03,'min_child_samples': 2400,'max_bins': 511,'seed': 42,'verbose': -1}\n\n# trained model with 100 iterations\nmodel = lgb.train(params, d_train, 100)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:03:45.813289Z","iopub.execute_input":"2022-12-12T15:03:45.813703Z","iopub.status.idle":"2022-12-12T15:05:43.441049Z","shell.execute_reply.started":"2022-12-12T15:03:45.813671Z","shell.execute_reply":"2022-12-12T15:05:43.439656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predications\n\npredictions = model.predict(test_dataset[num_columns])\npredictions","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:06:01.375501Z","iopub.execute_input":"2022-12-12T15:06:01.376060Z","iopub.status.idle":"2022-12-12T15:07:17.672737Z","shell.execute_reply.started":"2022-12-12T15:06:01.376010Z","shell.execute_reply":"2022-12-12T15:07:17.671777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Output\n\nsample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T15:07:29.975982Z","iopub.execute_input":"2022-12-12T15:07:29.976445Z","iopub.status.idle":"2022-12-12T15:07:35.864220Z","shell.execute_reply.started":"2022-12-12T15:07:29.976408Z","shell.execute_reply":"2022-12-12T15:07:35.863238Z"},"trusted":true},"execution_count":null,"outputs":[]}]}