{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport gc\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-16T19:53:19.726671Z","iopub.execute_input":"2022-08-16T19:53:19.727201Z","iopub.status.idle":"2022-08-16T19:53:20.412062Z","shell.execute_reply.started":"2022-08-16T19:53:19.727096Z","shell.execute_reply":"2022-08-16T19:53:20.411093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Trainning DataSet","metadata":{}},{"cell_type":"code","source":"train_dataset_ = pd.read_feather('../input/amexfeather/train_data.ftr')\n# Keep the latest statement features for each customer\ntrain_dataset = train_dataset_.groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:53:20.413737Z","iopub.execute_input":"2022-08-16T19:53:20.414399Z","iopub.status.idle":"2022-08-16T19:53:48.000892Z","shell.execute_reply.started":"2022-08-16T19:53:20.414363Z","shell.execute_reply":"2022-08-16T19:53:47.999011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset of this competition has a huge size. If you're reading raw CSV files, It will create a out of memory error. That's why we read the data from AMEX-Feather-Dataset.","metadata":{}},{"cell_type":"code","source":"del train_dataset_\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:53:48.003867Z","iopub.execute_input":"2022-08-16T19:53:48.004335Z","iopub.status.idle":"2022-08-16T19:53:48.233260Z","shell.execute_reply.started":"2022-08-16T19:53:48.004285Z","shell.execute_reply":"2022-08-16T19:53:48.232218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:53:48.235803Z","iopub.execute_input":"2022-08-16T19:53:48.236408Z","iopub.status.idle":"2022-08-16T19:53:48.278989Z","shell.execute_reply.started":"2022-08-16T19:53:48.236370Z","shell.execute_reply":"2022-08-16T19:53:48.277755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.info(max_cols=191,show_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:53:48.280205Z","iopub.execute_input":"2022-08-16T19:53:48.281250Z","iopub.status.idle":"2022-08-16T19:53:49.000033Z","shell.execute_reply.started":"2022-08-16T19:53:48.281212Z","shell.execute_reply":"2022-08-16T19:53:48.998783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:53:49.002112Z","iopub.execute_input":"2022-08-16T19:53:49.002601Z","iopub.status.idle":"2022-08-16T19:54:04.061964Z","shell.execute_reply.started":"2022-08-16T19:53:49.002553Z","shell.execute_reply":"2022-08-16T19:54:04.060752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore a Pattern","metadata":{}},{"cell_type":"markdown","source":"The dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:\n\n* D_* = Delinquency variables (bad or criminal behaviour, especially among young people)\n* S_* = Spend variables\n* P_* = Payment variables\n* B_* = Balance variables\n* R_* = Risk variables\n\nwith the following features being categorical: ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{}},{"cell_type":"code","source":"categorical_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nnum_cols = [col for col in train_dataset.columns if col not in categorical_cols + [\"target\"]]\n\nprint(f'Total number of features: {1}')\nprint(f'Total number of categorical features: {len(categorical_cols)}')\nprint(f'Total number of continuos features: {len(num_cols)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:54:04.063506Z","iopub.execute_input":"2022-08-16T19:54:04.064260Z","iopub.status.idle":"2022-08-16T19:54:04.072885Z","shell.execute_reply.started":"2022-08-16T19:54:04.064212Z","shell.execute_reply":"2022-08-16T19:54:04.071568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing Target","metadata":{}},{"cell_type":"code","source":"sns.countplot(x = 'target', data = train_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:54:04.074454Z","iopub.execute_input":"2022-08-16T19:54:04.074832Z","iopub.status.idle":"2022-08-16T19:54:04.418103Z","shell.execute_reply.started":"2022-08-16T19:54:04.074780Z","shell.execute_reply":"2022-08-16T19:54:04.417197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing categorial features","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20, 30))\nfor i, k in enumerate(categorical_cols):\n    plt.subplot(6, 2, i+1)\n    temp_val = pd.DataFrame(train_dataset[k].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp_val.index.name = 'value'\n    temp_val.reset_index(inplace=True)\n    plt.bar(temp_val.index, temp_val['count'], alpha=0.5)\n    plt.xlabel(k)\n    plt.ylabel('frequency')\n    plt.xticks(temp_val.index, temp_val.value)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:54:04.419438Z","iopub.execute_input":"2022-08-16T19:54:04.419948Z","iopub.status.idle":"2022-08-16T19:54:05.857575Z","shell.execute_reply.started":"2022-08-16T19:54:04.419915Z","shell.execute_reply":"2022-08-16T19:54:05.856164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing categorial features based on the target","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20, 30))\nfor i, f in enumerate(categorical_cols):\n    plt.subplot(6, 2, i+1)\n    temp = pd.DataFrame(train_dataset[f][train_dataset.target == 0].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp.index.name = 'value'\n    temp.reset_index(inplace=True)\n    plt.bar(temp.index, temp['count'], alpha=0.5, label='target=0')\n    temp = pd.DataFrame(train_dataset[f][train_dataset.target == 1].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp.index.name = 'value'\n    temp.reset_index(inplace=True)\n    plt.bar(temp.index, temp['count'], alpha=0.5, label='target=1')\n    plt.xlabel(f)\n    plt.ylabel('frequency')\n    plt.legend()\n    plt.xticks(temp.index, temp.value)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:54:05.863732Z","iopub.execute_input":"2022-08-16T19:54:05.864162Z","iopub.status.idle":"2022-08-16T19:54:08.127452Z","shell.execute_reply.started":"2022-08-16T19:54:05.864128Z","shell.execute_reply":"2022-08-16T19:54:08.126354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing continuous features","metadata":{}},{"cell_type":"code","source":"for i, l in enumerate(num_cols):\n    if i % 4 == 0: \n        if i > 0: plt.show()\n        plt.figure(figsize=(20, 3))\n    plt.subplot(1, 4, i % 4 + 1)\n    plt.hist(train_dataset[l], bins=200, color='#C69C73')\n    plt.xlabel(l)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:54:08.128646Z","iopub.execute_input":"2022-08-16T19:54:08.129721Z","iopub.status.idle":"2022-08-16T19:55:41.729455Z","shell.execute_reply.started":"2022-08-16T19:54:08.129666Z","shell.execute_reply":"2022-08-16T19:55:41.727943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Aggregated profile features","metadata":{}},{"cell_type":"code","source":"Delinquency = [d for d in train_dataset.columns if d.startswith('D_')]\nSpend = [s for s in train_dataset.columns if s.startswith('S_')]\nPayment = [p for p in train_dataset.columns if p.startswith('P_')]\nBalance = [b for b in train_dataset.columns if b.startswith('B_')]\nRisk = [r for r in train_dataset.columns if r.startswith('R_')]\nDict = {'Delinquency': len(Delinquency), 'Spend': len(Spend), 'Payment': len(Payment), 'Balance': len(Balance), 'Risk': len(Risk),}\n\nplt.figure(figsize=(10,5))\nsns.barplot(x=list(Dict.keys()), y=list(Dict.values()));","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:41.731086Z","iopub.execute_input":"2022-08-16T19:55:41.731449Z","iopub.status.idle":"2022-08-16T19:55:41.915548Z","shell.execute_reply.started":"2022-08-16T19:55:41.731416Z","shell.execute_reply":"2022-08-16T19:55:41.914099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check null values","metadata":{}},{"cell_type":"code","source":"NaN_Val = np.array(train_dataset.isnull().sum())\nNaN_prec = np.array((train_dataset.isnull().sum() * 100 / len(train_dataset)).round(2))\nNaN_Col = pd.DataFrame([np.array(list(train_dataset.columns)).T,NaN_Val.T,NaN_prec.T,np.array(list(train_dataset.dtypes)).T], index=['Features','Num of Missing values','Percentage','DataType']\n).transpose()\npd.set_option('display.max_rows', None)\nNaN_Col","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:41.917752Z","iopub.execute_input":"2022-08-16T19:55:41.918272Z","iopub.status.idle":"2022-08-16T19:55:42.858537Z","shell.execute_reply.started":"2022-08-16T19:55:41.918223Z","shell.execute_reply":"2022-08-16T19:55:42.857338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are many missing values in the dataset","metadata":{}},{"cell_type":"markdown","source":"# Drop unuseful columns","metadata":{}},{"cell_type":"markdown","source":"Remove columns if there are > 80% of missing values","metadata":{}},{"cell_type":"code","source":"train_dataset = train_dataset.drop(['S_2','D_66','D_42','D_49','D_73','D_76','R_9','B_29','D_87','D_88','D_106','R_26','D_108','D_110','D_111','B_39','B_42','D_132','D_134','D_135','D_136','D_137','D_138','D_142'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:42.860210Z","iopub.execute_input":"2022-08-16T19:55:42.860585Z","iopub.status.idle":"2022-08-16T19:55:43.168553Z","shell.execute_reply.started":"2022-08-16T19:55:42.860550Z","shell.execute_reply":"2022-08-16T19:55:43.167266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fill null values","metadata":{}},{"cell_type":"code","source":"selected_col = np.array(['P_2','S_3','B_2','D_41','D_43','B_3','D_44','D_45','D_46','D_48','D_50','D_53','S_7','D_56','S_9','B_6','B_8','D_52','P_3','D_54','D_55','B_13','D_59','D_61','B_15','D_62','B_16','B_17','D_77','B_19','B_20','D_69','B_22','D_70','D_72','D_74','R_7','B_25','B_26','D_78','D_79','D_80','B_27','D_81','R_12','D_82','D_105','S_27','D_83','R_14','D_84','D_86','R_20','B_33','D_89','D_91','S_22','S_23','S_24','S_25','S_26','D_102','D_103','D_104','D_107','B_37','R_27','D_109','D_112','B_40','D_113','D_115','D_118','D_119','D_121','D_122','D_123','D_124','D_125','D_128','D_129','B_41','D_130','D_131','D_133','D_139','D_140','D_141','D_143','D_144','D_145'])\n\nfor col in selected_col:\n    train_dataset[col] = train_dataset[col].fillna(train_dataset[col].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:43.171130Z","iopub.execute_input":"2022-08-16T19:55:43.172365Z","iopub.status.idle":"2022-08-16T19:55:45.026591Z","shell.execute_reply.started":"2022-08-16T19:55:43.172314Z","shell.execute_reply":"2022-08-16T19:55:45.025331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In describe session you saw, lot of cloumns means are NaN. So, that's why i have used median to fill NaN values. ","metadata":{}},{"cell_type":"code","source":"selcted_col2 = np.array(['D_68','B_30','B_38','D_64','D_114','D_116','D_117','D_120','D_126'])\n\nfor col2 in selcted_col2:\n    train_dataset[col2] =  train_dataset[col2].fillna(train_dataset[col2].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:45.027981Z","iopub.execute_input":"2022-08-16T19:55:45.028406Z","iopub.status.idle":"2022-08-16T19:55:45.087762Z","shell.execute_reply.started":"2022-08-16T19:55:45.028370Z","shell.execute_reply":"2022-08-16T19:55:45.086491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check again null values","metadata":{}},{"cell_type":"code","source":"print(train_dataset.isnull().sum().to_string())","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:45.088945Z","iopub.execute_input":"2022-08-16T19:55:45.089237Z","iopub.status.idle":"2022-08-16T19:55:45.448716Z","shell.execute_reply.started":"2022-08-16T19:55:45.089209Z","shell.execute_reply":"2022-08-16T19:55:45.447280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are no more missing values","metadata":{}},{"cell_type":"code","source":"train_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:45.450367Z","iopub.execute_input":"2022-08-16T19:55:45.450865Z","iopub.status.idle":"2022-08-16T19:55:45.459236Z","shell.execute_reply.started":"2022-08-16T19:55:45.450819Z","shell.execute_reply":"2022-08-16T19:55:45.458034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:45.460743Z","iopub.execute_input":"2022-08-16T19:55:45.461319Z","iopub.status.idle":"2022-08-16T19:55:45.491703Z","shell.execute_reply.started":"2022-08-16T19:55:45.461273Z","shell.execute_reply":"2022-08-16T19:55:45.490698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Testing DataSet","metadata":{}},{"cell_type":"code","source":"test_dataset_ = pd.read_feather('../input/amexfeather/test_data.ftr')\n# Keep the latest statement features for each customer\ntest_dataset = test_dataset_.groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:55:45.492922Z","iopub.execute_input":"2022-08-16T19:55:45.493335Z","iopub.status.idle":"2022-08-16T19:56:30.291624Z","shell.execute_reply.started":"2022-08-16T19:55:45.493302Z","shell.execute_reply":"2022-08-16T19:56:30.290260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_dataset_\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:30.293108Z","iopub.execute_input":"2022-08-16T19:56:30.293562Z","iopub.status.idle":"2022-08-16T19:56:30.490051Z","shell.execute_reply.started":"2022-08-16T19:56:30.293528Z","shell.execute_reply":"2022-08-16T19:56:30.488745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:30.491433Z","iopub.execute_input":"2022-08-16T19:56:30.491809Z","iopub.status.idle":"2022-08-16T19:56:30.525972Z","shell.execute_reply.started":"2022-08-16T19:56:30.491777Z","shell.execute_reply":"2022-08-16T19:56:30.524737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:30.528050Z","iopub.execute_input":"2022-08-16T19:56:30.528464Z","iopub.status.idle":"2022-08-16T19:56:30.539953Z","shell.execute_reply.started":"2022-08-16T19:56:30.528430Z","shell.execute_reply":"2022-08-16T19:56:30.538645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check null values","metadata":{}},{"cell_type":"code","source":"NaN_Val2 = np.array(test_dataset.isnull().sum())\nNaN_prec2 = np.array((test_dataset.isnull().sum() * 100 / len(test_dataset)).round(2))\nNaN_Col2 = pd.DataFrame([np.array(list(test_dataset.columns)).T,NaN_Val2.T,NaN_prec2.T,np.array(list(test_dataset.dtypes)).T], index=['Features','Num of Missing values','Percentage','DataType']\n).transpose()\npd.set_option('display.max_rows', None)\n\nNaN_Col2","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:30.541479Z","iopub.execute_input":"2022-08-16T19:56:30.542015Z","iopub.status.idle":"2022-08-16T19:56:32.409230Z","shell.execute_reply.started":"2022-08-16T19:56:30.541973Z","shell.execute_reply":"2022-08-16T19:56:32.407756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Drop unuseful columns","metadata":{}},{"cell_type":"code","source":"test_dataset = test_dataset.drop(['S_2','D_42','D_49','D_66','D_73','D_76','R_9','B_29','D_87','D_88','D_106','R_26','D_108','D_110','D_111','B_39','B_42','D_132','D_134','D_135','D_136','D_137','D_138','D_142'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:32.411051Z","iopub.execute_input":"2022-08-16T19:56:32.411452Z","iopub.status.idle":"2022-08-16T19:56:32.981672Z","shell.execute_reply.started":"2022-08-16T19:56:32.411418Z","shell.execute_reply":"2022-08-16T19:56:32.980209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fill null values","metadata":{}},{"cell_type":"code","source":"selected_column = np.array(['P_2','S_3','B_2','D_41','D_43','B_3','D_44','D_45','D_46','D_48','D_50','D_53','S_7','D_56','S_9','S_12','S_17','B_6','B_8','D_52','P_3','D_54','D_55','B_13','D_59','D_61','B_15','D_62','B_16','B_17','D_77','B_19','B_20','D_69','B_22','D_70','D_72','D_74','R_7','B_25','B_26','D_78','D_79','D_80','B_27','D_81','R_12','D_82','D_105','S_27','D_83','R_14','D_84','D_86','R_20','B_33','D_89','D_91','S_22','S_23','S_24','S_25','S_26','D_102','D_103','D_104','D_107','B_37','R_27','D_109','D_112','B_40','D_113','D_115','D_118','D_119','D_121','D_122','D_123','D_124','D_125','D_128','D_129','B_41','D_130','D_131','D_133','D_139','D_140','D_141','D_143','D_144','D_145'])\n\nfor column in selected_column:\n    test_dataset[column] = test_dataset[column].fillna(test_dataset[column].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:32.983370Z","iopub.execute_input":"2022-08-16T19:56:32.984145Z","iopub.status.idle":"2022-08-16T19:56:36.837010Z","shell.execute_reply.started":"2022-08-16T19:56:32.984085Z","shell.execute_reply":"2022-08-16T19:56:36.835851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_column2 = np.array(['D_68','B_30','B_38','D_114','D_116','D_117','D_120','D_126'])\n\nfor column2 in selected_column2:\n    test_dataset[column2] =  test_dataset[column2].fillna(test_dataset[column2].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:36.838403Z","iopub.execute_input":"2022-08-16T19:56:36.838837Z","iopub.status.idle":"2022-08-16T19:56:36.929284Z","shell.execute_reply.started":"2022-08-16T19:56:36.838799Z","shell.execute_reply":"2022-08-16T19:56:36.927763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check again null values","metadata":{}},{"cell_type":"code","source":"print(test_dataset.isnull().sum().to_string())","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:36.935958Z","iopub.execute_input":"2022-08-16T19:56:36.936373Z","iopub.status.idle":"2022-08-16T19:56:37.669756Z","shell.execute_reply.started":"2022-08-16T19:56:36.936336Z","shell.execute_reply":"2022-08-16T19:56:37.668639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:37.671304Z","iopub.execute_input":"2022-08-16T19:56:37.672314Z","iopub.status.idle":"2022-08-16T19:56:37.680363Z","shell.execute_reply.started":"2022-08-16T19:56:37.672255Z","shell.execute_reply":"2022-08-16T19:56:37.679010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:37.681850Z","iopub.execute_input":"2022-08-16T19:56:37.683024Z","iopub.status.idle":"2022-08-16T19:56:37.713579Z","shell.execute_reply.started":"2022-08-16T19:56:37.682979Z","shell.execute_reply":"2022-08-16T19:56:37.712461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert categorical variable to numbers","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\nenc = OrdinalEncoder()\ncategorical_cols.remove('D_66')\n\ntrain_dataset[categorical_cols] = enc.fit_transform(train_dataset[categorical_cols])\ntest_dataset[categorical_cols] = enc.transform(test_dataset[categorical_cols])","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:37.715469Z","iopub.execute_input":"2022-08-16T19:56:37.715941Z","iopub.status.idle":"2022-08-16T19:56:39.727123Z","shell.execute_reply.started":"2022-08-16T19:56:37.715896Z","shell.execute_reply":"2022-08-16T19:56:39.725962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Remove highly correlated features","metadata":{}},{"cell_type":"markdown","source":"Remove columns if there are > 90% of correlations","metadata":{}},{"cell_type":"code","source":"train_dataset_without_target = train_dataset.drop([\"target\"],axis=1)\n\ncor_matrix = train_dataset_without_target.corr()\ncol_core = set()\n\nfor i in range(len(cor_matrix.columns)):\n    for j in range(i):\n        if(cor_matrix.iloc[i, j] > 0.9):\n            col_name = cor_matrix.columns[i]\n            col_core.add(col_name)\ncol_core","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:56:39.728871Z","iopub.execute_input":"2022-08-16T19:56:39.729272Z","iopub.status.idle":"2022-08-16T19:57:14.528509Z","shell.execute_reply.started":"2022-08-16T19:56:39.729236Z","shell.execute_reply":"2022-08-16T19:57:14.527220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = train_dataset.drop(col_core, axis=1)\ntest_dataset = test_dataset.drop(col_core, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:57:14.530292Z","iopub.execute_input":"2022-08-16T19:57:14.531056Z","iopub.status.idle":"2022-08-16T19:57:15.384844Z","shell.execute_reply.started":"2022-08-16T19:57:14.531010Z","shell.execute_reply":"2022-08-16T19:57:15.383418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:57:15.386397Z","iopub.execute_input":"2022-08-16T19:57:15.386763Z","iopub.status.idle":"2022-08-16T19:57:15.394053Z","shell.execute_reply.started":"2022-08-16T19:57:15.386731Z","shell.execute_reply":"2022-08-16T19:57:15.392753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Model","metadata":{}},{"cell_type":"code","source":"num_columns = [col for col in train_dataset.columns if col not in [\"target\"]]\n\nX = train_dataset[num_columns]\ny = train_dataset['target']\n\nprint(f\"X shape is = {X.shape}\" )\nprint(f\"Y shape is = {y.shape}\" )","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:57:15.579232Z","iopub.execute_input":"2022-08-16T19:57:15.579600Z","iopub.status.idle":"2022-08-16T19:57:15.849967Z","shell.execute_reply.started":"2022-08-16T19:57:15.579567Z","shell.execute_reply":"2022-08-16T19:57:15.848905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nx_train,x_test,y_train,y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\nprint(f\"X_train shape is = {x_train.shape}\" )\nprint(f\"Y_train shape is = {y_train.shape}\" )\nprint(f\"X_test shape is = {x_test.shape}\" )\nprint(f\"Y_test shape is = {y_test.shape}\" )","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:57:15.851040Z","iopub.execute_input":"2022-08-16T19:57:15.851424Z","iopub.status.idle":"2022-08-16T19:57:17.088168Z","shell.execute_reply.started":"2022-08-16T19:57:15.851350Z","shell.execute_reply":"2022-08-16T19:57:17.086825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n\nd_train = lgb.Dataset(x_train, label=y_train, categorical_feature = categorical_cols)\n\nparams = {'objective': 'binary','n_estimators': 1200,'metric': 'binary_logloss','boosting': 'gbdt','num_leaves': 90,'reg_lambda' : 50,'colsample_bytree': 0.19,'learning_rate': 0.03,'min_child_samples': 2400,'max_bins': 511,'seed': 42,'verbose': -1}\n\n# trained model with 100 iterations\nmodel = lgb.train(params, d_train, 100)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:57:17.090400Z","iopub.execute_input":"2022-08-16T19:57:17.090761Z","iopub.status.idle":"2022-08-16T19:59:14.177457Z","shell.execute_reply.started":"2022-08-16T19:57:17.090728Z","shell.execute_reply":"2022-08-16T19:59:14.176338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Prediction","metadata":{}},{"cell_type":"code","source":"predictions = model.predict(test_dataset[num_columns])\npredictions","metadata":{"execution":{"iopub.status.busy":"2022-08-16T19:59:14.179146Z","iopub.execute_input":"2022-08-16T19:59:14.179511Z","iopub.status.idle":"2022-08-16T20:00:28.039951Z","shell.execute_reply.started":"2022-08-16T19:59:14.179477Z","shell.execute_reply":"2022-08-16T20:00:28.038656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Output","metadata":{}},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T20:00:28.041867Z","iopub.execute_input":"2022-08-16T20:00:28.042721Z","iopub.status.idle":"2022-08-16T20:00:33.065369Z","shell.execute_reply.started":"2022-08-16T20:00:28.042655Z","shell.execute_reply":"2022-08-16T20:00:33.064033Z"},"trusted":true},"execution_count":null,"outputs":[]}]}