{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Introduction**","metadata":{}},{"cell_type":"markdown","source":"This notebook is intended to be a starter for anyone who wants to make attempt to this challenge.  \n\nMany people are struggling to manage with datasets or they are relying on datasets created by other Kagglers. Some also try to move datasets to ther cloud services to have more compute, memory and storage.\n\n**I wanted to have a noebook that we can run on Kaggle only without need of any other cloud service or datasets.**\n\nHere is a humble attempt. I request my fellow Kagglers to give suggestions and improve performance.","metadata":{}},{"cell_type":"markdown","source":"# **How to Use This Notebook**","metadata":{}},{"cell_type":"markdown","source":"If you are running this notebook for very first time, then follow below steps:-\n* Uncomment preprocessing steps for level 1.\n* Save Version with \"Save & Run All\" option. This will run notebook in background.\n* Once you receive notification, output folder will have processed training and test data.\n* In case your session is reset and you lose files in output folder, you can add notbook output datasets using \"Add Data\" option.\n\n# **If you are using supporting datasets for this notebook then start from section \"Train Model\".**\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# **Setup**","metadata":{}},{"cell_type":"code","source":"import vaex\nvaex.multithreading.thread_count_default = 8\nimport vaex.ml\n\nimport pandas as pd\nimport numpy  as np \n\nimport os\nimport gc\nimport psutil\nimport glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-20T18:00:39.281435Z","iopub.execute_input":"2022-07-20T18:00:39.281871Z","iopub.status.idle":"2022-07-20T18:00:39.289084Z","shell.execute_reply.started":"2022-07-20T18:00:39.281839Z","shell.execute_reply":"2022-07-20T18:00:39.287630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Utility Functions**","metadata":{}},{"cell_type":"code","source":"\ndef remove_output_files(file_pattern):\n    fileList = glob.glob(file_pattern)\n    for filePath in fileList:\n        try:\n            os.remove(filePath)\n        except:\n            print(\"Error while deleting file : \", filePath)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.298388Z","iopub.execute_input":"2022-07-20T18:00:39.298834Z","iopub.status.idle":"2022-07-20T18:00:39.306393Z","shell.execute_reply.started":"2022-07-20T18:00:39.298792Z","shell.execute_reply":"2022-07-20T18:00:39.305017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_and_convert_floats(ddf):\n    for c in ddf.columns:\n        if ddf[c].dtype == 'float64':\n            ddf[c] = ddf[c].fillna(0.0).astype('float32')\n    return ddf\n\ndef encode_cat_features(df):\n    cat_features = ['D_63','D_64']\n    label_encoder = vaex.ml.LabelEncoder(features=cat_features)\n    df = label_encoder.fit_transform(df)\n    df.drop(cat_features, inplace=True)\n    df.rename('label_encoded_D_63','D_63')\n    df.rename('label_encoded_D_64','D_64')\n    df['D_64'] = df['D_64'].astype('float32')\n    df['D_63'] = df['D_63'].astype('float32')\n    df['B_31'] = df['B_31'].astype('float32')\n    return df\n\ndef get_last_statement(df):\n    return df.groupby(['customer_ID']).agg({col: vaex.agg.last(col) for col in df.get_column_names() if col not in [\"customer_ID\"]})\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.318098Z","iopub.execute_input":"2022-07-20T18:00:39.318966Z","iopub.status.idle":"2022-07-20T18:00:39.331187Z","shell.execute_reply.started":"2022-07-20T18:00:39.318924Z","shell.execute_reply":"2022-07-20T18:00:39.330350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_last_statement_ex(df_test):\n    delinquency_features = [col for col in df_test if col.startswith('D_')] \n    df = df_test.groupby(['customer_ID']).agg({col: vaex.agg.last(col) for col in df_test.get_column_names() if col not in delinquency_features + [\"customer_ID\"]})\n    df.export_hdf5('./last-statement-p1.hdf5')\n    del df\n    gc.collect()\n    delinquency_features = ['S_2'] + [col for col in df_test if col.startswith('D_') and len(col) == 4] \n    delinquency_features2 = ['S_2'] + [col for col in df_test if col.startswith('D_') and len(col) == 5]\n    df_2 = df_test.groupby(['customer_ID']).agg({col: vaex.agg.last(col) for col in df_test.get_column_names() if col in delinquency_features})\n    df_2.export_hdf5('./last-statement-p2.hdf5')\n    del df_2\n    gc.collect()\n    df_3 = df_test.groupby(['customer_ID']).agg({col: vaex.agg.last(col) for col in df_test.get_column_names() if col in delinquency_features2})\n    df_3.export_hdf5('./last-statement-p3.hdf5')\n    del df_3\n    gc.collect()\n    last_statement_p1 = vaex.open('./last-statement-p1.hdf5')\n    last_statement_p2 = vaex.open('./last-statement-p2.hdf5')\n    last_statement_p3 = vaex.open('./last-statement-p3.hdf5')\n    last_statement_p1 = last_statement_p1.drop('S_2')\n    last_statement_p2 = last_statement_p2.drop('S_2')\n    last_statement_p3 = last_statement_p3.drop('S_2')\n    gc.collect()\n    last_statement_p1 = last_statement_p1.join(last_statement_p2, how=\"inner\", on='customer_ID')\n    df_last_statement = last_statement_p1.join(last_statement_p3, how=\"inner\", on='customer_ID')\n    del last_statement_p1\n    del last_statement_p2\n    del last_statement_p3\n    gc.collect()\n    statement_path = './last-statement-p*.hdf5'\n    remove_output_files(statement_path)\n    return df_last_statement","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.337432Z","iopub.execute_input":"2022-07-20T18:00:39.338075Z","iopub.status.idle":"2022-07-20T18:00:39.353266Z","shell.execute_reply.started":"2022-07-20T18:00:39.338041Z","shell.execute_reply":"2022-07-20T18:00:39.352292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data(data):\n    for i, df in enumerate(vaex.from_csv(f'../input/amex-default-prediction/{data}.csv', chunk_size=500_000)):\n        df['S_2'] = df['S_2'].str.replace('-','').astype('float32')\n        #df['R_26'] = df['R_26'].astype('int16')\n        df = fill_and_convert_floats(df)\n        df = encode_cat_features(df)\n        df = get_last_statement(df)\n        export_path = f'./{data}_{i:02}.hdf5'    \n        df.export_hdf5(export_path)\n        del df\n        gc.collect()\n    import_path = f'./{data}_*.hdf5'\n    df = vaex.open(import_path)\n    df.export_hdf5(f'./{data}.hdf5')\n    del df\n    gc.collect()\n    remove_output_files(import_path)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.354716Z","iopub.execute_input":"2022-07-20T18:00:39.355356Z","iopub.status.idle":"2022-07-20T18:00:39.369551Z","shell.execute_reply.started":"2022-07-20T18:00:39.355321Z","shell.execute_reply":"2022-07-20T18:00:39.367997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data_level2(df, data, flag):\n    if flag == 1:\n        df = get_last_statement_ex(df)     \n    else:\n        df = get_last_statement(df)\n        df.drop('S_2', inplace=True)   \n    label_encoder = vaex.ml.LabelEncoder(features=['customer_ID'])\n    df = label_encoder.fit_transform(df)\n    df_customer_map = df[['label_encoded_customer_ID', 'customer_ID']]\n    df.drop('customer_ID', inplace=True)\n    df.rename('label_encoded_customer_ID','customer_ID')\n    df_customer_map.export_hdf5(f'./{data}_customer_map.hdf5')\n    df.export_hdf5(f'./{data}v2.hdf5')\n    del df\n    del df_customer_map\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.380461Z","iopub.execute_input":"2022-07-20T18:00:39.381670Z","iopub.status.idle":"2022-07-20T18:00:39.390107Z","shell.execute_reply.started":"2022-07-20T18:00:39.381625Z","shell.execute_reply":"2022-07-20T18:00:39.389100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.454018Z","iopub.execute_input":"2022-07-20T18:00:39.455007Z","iopub.status.idle":"2022-07-20T18:00:39.470542Z","shell.execute_reply.started":"2022-07-20T18:00:39.454964Z","shell.execute_reply":"2022-07-20T18:00:39.469184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric_np(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\ndef lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric_np(y_true, y_pred), True","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.477426Z","iopub.execute_input":"2022-07-20T18:00:39.477844Z","iopub.status.idle":"2022-07-20T18:00:39.491361Z","shell.execute_reply.started":"2022-07-20T18:00:39.477792Z","shell.execute_reply":"2022-07-20T18:00:39.490166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preprocessing - Level 1 - Reduce Train and Test Datasets**","metadata":{}},{"cell_type":"markdown","source":"In this step train and test datasets will be processed to reduce the size so we can perform feature engineering and build models without worrying about memory, CPU and hard disk constraints.\n\nUncomment below lines only if you need to create reduced version of data. **If size reduced data is already availlable then skip these steps.**\n\nBelow changes are performed on datasets:-\n* S_2 date feature is changed to float32\n* All float64 features are converted to float32\n* Missing values in all numeric features are set to 0.0\n* Categorical features D_63 and D_64 are encoded - This was required to extract last statement\n* Categorical features B_31 is converted to float32 - This was required to extract last statement\n* For each customer only last statement is kept as available in each chunk\n","metadata":{}},{"cell_type":"markdown","source":"**Uncomment below two lines to generate level 1 data. Alternatively you can use this [dataset](https://www.kaggle.com/datasets/mirfanazam/amex-prediction-starter-level-1).**","metadata":{}},{"cell_type":"code","source":"# process_data('train_data')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.496577Z","iopub.execute_input":"2022-07-20T18:00:39.497149Z","iopub.status.idle":"2022-07-20T18:00:39.506474Z","shell.execute_reply.started":"2022-07-20T18:00:39.497117Z","shell.execute_reply":"2022-07-20T18:00:39.505313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# process_data('test_data')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.517903Z","iopub.execute_input":"2022-07-20T18:00:39.518693Z","iopub.status.idle":"2022-07-20T18:00:39.524065Z","shell.execute_reply.started":"2022-07-20T18:00:39.518651Z","shell.execute_reply":"2022-07-20T18:00:39.522634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Read Train and Test Data Preprocessed at Level 1**","metadata":{}},{"cell_type":"markdown","source":"Use below lines to use train and test data from output folder. This is required when you want to run this notebook as a whole and process data, train model, make prediction and make submission.","metadata":{}},{"cell_type":"code","source":"# df_train = vaex.open('./train_data.hdf5')\n# df_test = vaex.open('./test_data.hdf5')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.530378Z","iopub.execute_input":"2022-07-20T18:00:39.530951Z","iopub.status.idle":"2022-07-20T18:00:39.535538Z","shell.execute_reply.started":"2022-07-20T18:00:39.530920Z","shell.execute_reply":"2022-07-20T18:00:39.534369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use below lines if you want to load level 1 data from input folder.","metadata":{}},{"cell_type":"code","source":"# df_train = vaex.open('../input/amex-prediction-starter-level-1/train_data.hdf5')\n# df_test = vaex.open('../input/amex-prediction-starter-level-1/test_data.hdf5')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.545256Z","iopub.execute_input":"2022-07-20T18:00:39.545850Z","iopub.status.idle":"2022-07-20T18:00:39.550748Z","shell.execute_reply.started":"2022-07-20T18:00:39.545817Z","shell.execute_reply":"2022-07-20T18:00:39.549082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preprocessing - Level 2**","metadata":{}},{"cell_type":"markdown","source":"In this step both training and test datasets will be processed: -\n\n* Keep only last statement for each customer\n* Remove date feature S_2\n* Encode Customer ID\n","metadata":{}},{"cell_type":"markdown","source":"**Uncomment below two lines to generate level 2 data. Alternatively you can use this [dataset](https://www.kaggle.com/datasets/mirfanazam/amex-prediction-starter-level-2).**","metadata":{}},{"cell_type":"code","source":"# process_data_level2(df_train, 'train_data', 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.558156Z","iopub.execute_input":"2022-07-20T18:00:39.559014Z","iopub.status.idle":"2022-07-20T18:00:39.564127Z","shell.execute_reply.started":"2022-07-20T18:00:39.558971Z","shell.execute_reply":"2022-07-20T18:00:39.562818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# process_data_level2(df_test, 'test_data', 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.573877Z","iopub.execute_input":"2022-07-20T18:00:39.574516Z","iopub.status.idle":"2022-07-20T18:00:39.583527Z","shell.execute_reply.started":"2022-07-20T18:00:39.574457Z","shell.execute_reply":"2022-07-20T18:00:39.579923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Read Train and Test Data Preprocessed at Level 2**","metadata":{}},{"cell_type":"markdown","source":"Use below lines to use train and test data from output folder. This is required when you want to run this notebook as a whole and process data, train model, make prediction and make submission.","metadata":{}},{"cell_type":"code","source":"# df_train = vaex.open('./train_datav2.hdf5')\n# df_test = vaex.open('./test_datav2.hdf5')\n# df_train_map = vaex.open('./train_data_customer_map.hdf5')\n# df_test_map = vaex.open('./test_data_customer_map.hdf5')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.586725Z","iopub.execute_input":"2022-07-20T18:00:39.587903Z","iopub.status.idle":"2022-07-20T18:00:39.597871Z","shell.execute_reply.started":"2022-07-20T18:00:39.587860Z","shell.execute_reply":"2022-07-20T18:00:39.593458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use below lines if you want to load level 2 data from input folder.","metadata":{}},{"cell_type":"code","source":"df_train = vaex.open('../input/amex-prediction-starter-level-2/train_datav2.hdf5')\ndf_test = vaex.open('../input/amex-prediction-starter-level-2/test_datav2.hdf5')\ndf_train_map = vaex.open('../input/amex-prediction-starter-level-2/train_data_customer_map.hdf5')\ndf_test_map = vaex.open('../input/amex-prediction-starter-level-2/test_data_customer_map.hdf5')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:39.604519Z","iopub.execute_input":"2022-07-20T18:00:39.604935Z","iopub.status.idle":"2022-07-20T18:00:40.605756Z","shell.execute_reply.started":"2022-07-20T18:00:39.604888Z","shell.execute_reply":"2022-07-20T18:00:40.604296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prepare Data to Train Model**","metadata":{}},{"cell_type":"code","source":"df_train_labels = vaex.open('../input/amex-default-prediction/train_labels.csv')\ndf_train_labels = df_train_labels.join(df_train_map, how=\"inner\", on=\"customer_ID\")\n\nall_features = [col for col in df_train]\n\ndf_customer = df_train[all_features]\ndf_customer = df_customer.join(df_train_labels, left_on='customer_ID', right_on='label_encoded_customer_ID', how='inner')\ndf_customer.drop(['label_encoded_customer_ID'], inplace=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-20T18:00:40.607964Z","iopub.execute_input":"2022-07-20T18:00:40.608373Z","iopub.status.idle":"2022-07-20T18:00:43.269474Z","shell.execute_reply.started":"2022-07-20T18:00:40.608337Z","shell.execute_reply":"2022-07-20T18:00:43.268007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Convert Vaex DataFrames to Pandas DataFrames**","metadata":{}},{"cell_type":"markdown","source":"**There is not much information available for Veax ML wrappers. So we will convert our dataframes from Vaex to Pandas.**","metadata":{}},{"cell_type":"code","source":"df_train = df_train.to_pandas_df()\ndf_test = df_test.to_pandas_df()\ndf_train_map = df_train_map.to_pandas_df()\ndf_test_map = df_test_map.to_pandas_df()\n\ndf_train_labels = df_train_labels.to_pandas_df()\ndf_customer = df_customer.to_pandas_df()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:43.271329Z","iopub.execute_input":"2022-07-20T18:00:43.271805Z","iopub.status.idle":"2022-07-20T18:00:49.201049Z","shell.execute_reply.started":"2022-07-20T18:00:43.271757Z","shell.execute_reply":"2022-07-20T18:00:49.199674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Train Model**","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom lightgbm import log_evaluation\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:49.204338Z","iopub.execute_input":"2022-07-20T18:00:49.204841Z","iopub.status.idle":"2022-07-20T18:00:49.210921Z","shell.execute_reply.started":"2022-07-20T18:00:49.204806Z","shell.execute_reply":"2022-07-20T18:00:49.209282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df_customer.pop('target')\n# c = df_customer.pop('customer_ID')\n# model_features = [col for col in df_customer if col not in [\"customer_ID\"]]\nmodel_features = [col for col in df_customer]\nX = df_customer[model_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:49.213461Z","iopub.execute_input":"2022-07-20T18:00:49.213974Z","iopub.status.idle":"2022-07-20T18:00:49.303966Z","shell.execute_reply.started":"2022-07-20T18:00:49.213942Z","shell.execute_reply":"2022-07-20T18:00:49.302671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X.drop(columns={'D_139', 'D_103'},inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:49.305652Z","iopub.execute_input":"2022-07-20T18:00:49.306158Z","iopub.status.idle":"2022-07-20T18:00:49.388825Z","shell.execute_reply.started":"2022-07-20T18:00:49.306106Z","shell.execute_reply":"2022-07-20T18:00:49.387765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:49.390243Z","iopub.execute_input":"2022-07-20T18:00:49.390803Z","iopub.status.idle":"2022-07-20T18:00:49.397527Z","shell.execute_reply.started":"2022-07-20T18:00:49.390771Z","shell.execute_reply":"2022-07-20T18:00:49.396303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:49.398781Z","iopub.execute_input":"2022-07-20T18:00:49.399725Z","iopub.status.idle":"2022-07-20T18:00:49.956586Z","shell.execute_reply.started":"2022-07-20T18:00:49.399688Z","shell.execute_reply":"2022-07-20T18:00:49.955681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtrain = lgb.Dataset(\n    data=X_train,\n    label=y_train\n)\n\ndvalid = lgb.Dataset(\n    data=X_test,\n    label=y_test,\n    reference=dtrain\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:49.957821Z","iopub.execute_input":"2022-07-20T18:00:49.958325Z","iopub.status.idle":"2022-07-20T18:00:49.963525Z","shell.execute_reply.started":"2022-07-20T18:00:49.958293Z","shell.execute_reply":"2022-07-20T18:00:49.962536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ndf_customer['D_117'] = df_customer['D_117'] + 1\ndf_customer['D_126'] = df_customer['D_126'] + 1","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:49.966977Z","iopub.execute_input":"2022-07-20T18:00:49.967329Z","iopub.status.idle":"2022-07-20T18:00:49.979817Z","shell.execute_reply.started":"2022-07-20T18:00:49.967290Z","shell.execute_reply":"2022-07-20T18:00:49.978536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_customer[categorical_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:00:49.981063Z","iopub.execute_input":"2022-07-20T18:00:49.981621Z","iopub.status.idle":"2022-07-20T18:00:49.986061Z","shell.execute_reply.started":"2022-07-20T18:00:49.981590Z","shell.execute_reply":"2022-07-20T18:00:49.985356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_params={\n    \"objective\": \"binary\",\n    \"n_estimators\": 1200,\n    \"learning_rate\": 0.03,\n    \"reg_lambda\": 50,\n    \"min_child_samples\": 2400,\n    \"num_leaves\": 220,\n    \"colsample_bytree\": 0.19,\n#     device='gpu',\n    \"random_state\": 1,\n    'verbose': -1\n}\n    \nmodel = lgb.train(\n    params=lgb_params,\n    train_set=dtrain,\n    valid_sets=[dvalid],\n    feval=lgb_amex_metric,\n    callbacks=[log_evaluation(100)]\n)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-20T18:16:06.524671Z","iopub.execute_input":"2022-07-20T18:16:06.525142Z","iopub.status.idle":"2022-07-20T18:18:31.886468Z","shell.execute_reply.started":"2022-07-20T18:16:06.525107Z","shell.execute_reply":"2022-07-20T18:18:31.885274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:03:59.370379Z","iopub.execute_input":"2022-07-20T18:03:59.370782Z","iopub.status.idle":"2022-07-20T18:03:59.787955Z","shell.execute_reply.started":"2022-07-20T18:03:59.370748Z","shell.execute_reply":"2022-07-20T18:03:59.786727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:04:02.070644Z","iopub.execute_input":"2022-07-20T18:04:02.071024Z","iopub.status.idle":"2022-07-20T18:04:07.805537Z","shell.execute_reply.started":"2022-07-20T18:04:02.070995Z","shell.execute_reply":"2022-07-20T18:04:07.804341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:49:17.984594Z","iopub.execute_input":"2022-07-18T13:49:17.987080Z","iopub.status.idle":"2022-07-18T13:49:17.997356Z","shell.execute_reply.started":"2022-07-18T13:49:17.987028Z","shell.execute_reply":"2022-07-18T13:49:17.995570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Make Prediction**","metadata":{}},{"cell_type":"code","source":"model_features = [col for col in df_customer]\ndf_customer_test = df_test[model_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:04:39.528822Z","iopub.execute_input":"2022-07-20T18:04:39.529196Z","iopub.status.idle":"2022-07-20T18:04:39.749176Z","shell.execute_reply.started":"2022-07-20T18:04:39.529166Z","shell.execute_reply":"2022-07-20T18:04:39.747802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_customer_test = df_customer_test.drop(columns={'D_139', 'D_103'})","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:04:48.286824Z","iopub.execute_input":"2022-07-20T18:04:48.288190Z","iopub.status.idle":"2022-07-20T18:04:48.509988Z","shell.execute_reply.started":"2022-07-20T18:04:48.288120Z","shell.execute_reply":"2022-07-20T18:04:48.508756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_test_pred = model.predict(df_customer_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:04:53.264380Z","iopub.execute_input":"2022-07-20T18:04:53.264779Z","iopub.status.idle":"2022-07-20T18:05:47.165438Z","shell.execute_reply.started":"2022-07-20T18:04:53.264747Z","shell.execute_reply":"2022-07-20T18:05:47.164385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_test_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:07:07.837600Z","iopub.execute_input":"2022-07-20T18:07:07.838053Z","iopub.status.idle":"2022-07-20T18:07:07.846294Z","shell.execute_reply.started":"2022-07-20T18:07:07.838005Z","shell.execute_reply":"2022-07-20T18:07:07.845344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Make Submission**","metadata":{}},{"cell_type":"code","source":"df_customer_test","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:56:12.505770Z","iopub.execute_input":"2022-07-18T13:56:12.506198Z","iopub.status.idle":"2022-07-18T13:56:12.622164Z","shell.execute_reply.started":"2022-07-18T13:56:12.506163Z","shell.execute_reply":"2022-07-18T13:56:12.620755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_test = pd.merge(df_customer_test, df_test_map, how=\"inner\", left_on=\"customer_ID\", right_on=\"label_encoded_customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:56:19.777519Z","iopub.execute_input":"2022-07-18T13:56:19.778105Z","iopub.status.idle":"2022-07-18T13:56:20.383193Z","shell.execute_reply.started":"2022-07-18T13:56:19.778048Z","shell.execute_reply":"2022-07-18T13:56:20.382037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_test = df_customer_test.drop(columns={'customer_ID_x','label_encoded_customer_ID'})","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:56:25.167177Z","iopub.execute_input":"2022-07-18T13:56:25.167678Z","iopub.status.idle":"2022-07-18T13:56:25.433097Z","shell.execute_reply.started":"2022-07-18T13:56:25.167629Z","shell.execute_reply":"2022-07-18T13:56:25.431620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_test_pred = pd.DataFrame(df_customer_test_pred.tolist())","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:56:27.266082Z","iopub.execute_input":"2022-07-18T13:56:27.266544Z","iopub.status.idle":"2022-07-18T13:56:27.484893Z","shell.execute_reply.started":"2022-07-18T13:56:27.266510Z","shell.execute_reply":"2022-07-18T13:56:27.483557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_test['prediction'] = df_customer_test_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:56:29.568325Z","iopub.execute_input":"2022-07-18T13:56:29.568784Z","iopub.status.idle":"2022-07-18T13:56:29.582147Z","shell.execute_reply.started":"2022-07-18T13:56:29.568750Z","shell.execute_reply":"2022-07-18T13:56:29.580802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_test","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:56:31.872539Z","iopub.execute_input":"2022-07-18T13:56:31.873969Z","iopub.status.idle":"2022-07-18T13:56:32.153519Z","shell.execute_reply.started":"2022-07-18T13:56:31.873923Z","shell.execute_reply":"2022-07-18T13:56:32.152103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_test = df_customer_test.rename(columns={\"customer_ID_y\": \"customer_ID\"})","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:56:38.830301Z","iopub.execute_input":"2022-07-18T13:56:38.830775Z","iopub.status.idle":"2022-07-18T13:56:39.156128Z","shell.execute_reply.started":"2022-07-18T13:56:38.830739Z","shell.execute_reply":"2022-07-18T13:56:39.154999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_prediction = df_customer_test[[\"customer_ID\", \"prediction\"]]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:56:42.441779Z","iopub.execute_input":"2022-07-18T13:56:42.442844Z","iopub.status.idle":"2022-07-18T13:56:42.475797Z","shell.execute_reply.started":"2022-07-18T13:56:42.442799Z","shell.execute_reply":"2022-07-18T13:56:42.474392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_prediction.to_csv(\"./submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T00:38:53.641810Z","iopub.execute_input":"2022-07-17T00:38:53.642776Z","iopub.status.idle":"2022-07-17T00:38:58.079574Z","shell.execute_reply.started":"2022-07-17T00:38:53.642728Z","shell.execute_reply":"2022-07-17T00:38:58.078286Z"},"trusted":true},"execution_count":null,"outputs":[]}]}