{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-13T05:32:20.635235Z","iopub.execute_input":"2023-04-13T05:32:20.636754Z","iopub.status.idle":"2023-04-13T05:32:20.646147Z","shell.execute_reply.started":"2023-04-13T05:32:20.636703Z","shell.execute_reply":"2023-04-13T05:32:20.644370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import package here\nimport sys\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder\nimport missingno as msno\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:20.648301Z","iopub.execute_input":"2023-04-13T05:32:20.649078Z","iopub.status.idle":"2023-04-13T05:32:20.658739Z","shell.execute_reply.started":"2023-04-13T05:32:20.649029Z","shell.execute_reply":"2023-04-13T05:32:20.657438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\n\ninput_path = Path('/kaggle/input/amex-default-prediction/')","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:20.660757Z","iopub.execute_input":"2023-04-13T05:32:20.661215Z","iopub.status.idle":"2023-04-13T05:32:20.674492Z","shell.execute_reply.started":"2023-04-13T05:32:20.661178Z","shell.execute_reply":"2023-04-13T05:32:20.672877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading dataset train_data.csv\ntrain_df_sample = pd.read_csv('../input/amex-default-prediction/train_data.csv', nrows=100000)\n# Loading dataset train_labels.csv\ntrain_label_df = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n# Loading dataset test_data.csv\ntest_df = pd.read_csv('../input/amex-default-prediction/test_data.csv', nrows=100000, index_col='customer_ID')\n# Merge of train_df_sample and train_label_df dataframe using key as customer_ID\ntrain_df = pd.merge(train_df_sample, train_label_df, how=\"inner\", on=[\"customer_ID\"])\ncategorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:20.678871Z","iopub.execute_input":"2023-04-13T05:32:20.679486Z","iopub.status.idle":"2023-04-13T05:32:28.333754Z","shell.execute_reply.started":"2023-04-13T05:32:20.679400Z","shell.execute_reply":"2023-04-13T05:32:28.332603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The info method finds that there are objects in the table besides numbers, which needs to be processed. Since there are many parameters, the missing data can be visualized by visualization.","metadata":{}},{"cell_type":"code","source":"# Data Summary\nsummary = train_df.describe(include='all').T\nsummary['missing'] = train_df.isnull().sum()\nsummary['unique'] = train_df.nunique()\nsummary['type'] = train_df.dtypes\n\n# Print\nprint(summary)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:29.922689Z","iopub.execute_input":"2023-04-13T05:32:29.923676Z","iopub.status.idle":"2023-04-13T05:32:32.420976Z","shell.execute_reply.started":"2023-04-13T05:32:29.923609Z","shell.execute_reply":"2023-04-13T05:32:32.419588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Features are anonymized and normalized, and fall into the following general categories:\n\nD_* = Delinquency variables\nS_* = Spend variables\nP_* = Payment variables\nB_* = Balance variables\nR_* = Risk variables","metadata":{}},{"cell_type":"markdown","source":"Show data distribution for different general categories:","metadata":{}},{"cell_type":"code","source":"# Group according to prefix\nfeature_groups = {}\nfor feature in train_df.columns.tolist():\n    prefix = feature[0]\n    if prefix not in feature_groups:\n        feature_groups[prefix] = []\n    feature_groups[prefix].append(feature)\n\n# Print\nfor prefix, group in feature_groups.items():\n    print(f\"Feature group {prefix}: {group}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:32.422629Z","iopub.execute_input":"2023-04-13T05:32:32.422997Z","iopub.status.idle":"2023-04-13T05:32:32.432221Z","shell.execute_reply.started":"2023-04-13T05:32:32.422961Z","shell.execute_reply":"2023-04-13T05:32:32.430660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation between target and all the other columns\ncorrelations = train_df.corrwith(train_df['target'])\n\n# Print results\nprint(\"Correlations between target and other columns:\")\nprint(correlations)\n\n# Find columns with correlation greater than 0.5\nhigh_correlations = correlations[abs(correlations) > 0.5]\n\n# Print results\nprint(\"\\n\")\nprint(\"strongly correlated values with 'target' (with correlation greater than 0.5):\")\nprint(high_correlations)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:32.433886Z","iopub.execute_input":"2023-04-13T05:32:32.434278Z","iopub.status.idle":"2023-04-13T05:32:32.818311Z","shell.execute_reply.started":"2023-04-13T05:32:32.434243Z","shell.execute_reply":"2023-04-13T05:32:32.816768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show distribution of 'target' using histplot\nsns.histplot(train_df['target'], color='g', bins=100, alpha=0.4)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:32.822346Z","iopub.execute_input":"2023-04-13T05:32:32.822767Z","iopub.status.idle":"2023-04-13T05:32:33.351006Z","shell.execute_reply.started":"2023-04-13T05:32:32.822730Z","shell.execute_reply":"2023-04-13T05:32:33.349607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = train_df.drop('target', axis=1).corr()\nplt.figure(figsize=(50, 50))\n\nsns.heatmap(corr[(abs(corr) >= 0.5) | (abs(corr) <= -0.4)], \n            cmap='viridis', vmax=1.0, vmin=-1.0, linewidths=0.1,\n            annot=True, annot_kws={\"size\": 8}, square=True);\nplt.savefig(\"heatmap.png\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:33.352872Z","iopub.execute_input":"2023-04-13T05:32:33.354090Z","iopub.status.idle":"2023-04-13T05:32:55.976323Z","shell.execute_reply.started":"2023-04-13T05:32:33.354033Z","shell.execute_reply":"2023-04-13T05:32:55.974829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select numerical features\nnum_df = train_df.select_dtypes(exclude='object')\n# Plot distributions of numerical features\nncols = 7\nnrows = 28\nfig, axes = plt.subplots(nrows, ncols, figsize=(4*ncols, 4*nrows), sharex=False)\n\nfor idx, ax in enumerate(axes.ravel()):\n    if idx < len(num_df.columns):\n        column = num_df.columns[idx]\n        ax.hist(num_df[column].dropna(), bins=50)\n        ax.set_title(column)\n        ax.set_ylabel(\"Frequency\")\n    else:\n        ax.set_axis_off()\n\n# Save the histograms as an image file\nplt.savefig(\"histograms.png\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:32:55.977884Z","iopub.execute_input":"2023-04-13T05:32:55.978228Z","iopub.status.idle":"2023-04-13T05:33:51.136051Z","shell.execute_reply.started":"2023-04-13T05:32:55.978194Z","shell.execute_reply":"2023-04-13T05:33:51.134548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cal_df = train_df[categorical_features]\n# Plot countplots for the selected features\nfig, axes = plt.subplots(round(len(cal_df.columns) / 3), 3, figsize=(12, 12))\n\nfor i, ax in enumerate(fig.axes):\n    if i < len(cal_df.columns):\n        sns.countplot(x=cal_df.columns[i], alpha=0.7, data=cal_df, ax=ax)\n        ax.set_xticklabels(ax.get_xticklabels(), rotation=45)\n\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:33:51.137976Z","iopub.execute_input":"2023-04-13T05:33:51.138887Z","iopub.status.idle":"2023-04-13T05:33:52.810193Z","shell.execute_reply.started":"2023-04-13T05:33:51.138840Z","shell.execute_reply":"2023-04-13T05:33:52.808622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Statistical missing value:","metadata":{}},{"cell_type":"code","source":"# The amount of missing values for each column\nmissing_values = train_df.isnull().sum()\n\n# Display the first 5 rows\nprint(missing_values.head(5))\n\n# Display an empty line for separation\nprint(\"...\")\n\n# Display the last 5 rows\nprint(missing_values.tail(5))","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:33:52.811822Z","iopub.execute_input":"2023-04-13T05:33:52.812157Z","iopub.status.idle":"2023-04-13T05:33:52.887364Z","shell.execute_reply.started":"2023-04-13T05:33:52.812124Z","shell.execute_reply":"2023-04-13T05:33:52.885858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visulize missing values:","metadata":{}},{"cell_type":"code","source":"# Randomly get 1919 sample data\ntrain_df_sample = train_df.sample(1919)  # 随机取样1919个数据\nmsno.matrix(train_df_sample)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:33:52.889293Z","iopub.execute_input":"2023-04-13T05:33:52.889844Z","iopub.status.idle":"2023-04-13T05:33:53.856846Z","shell.execute_reply.started":"2023-04-13T05:33:52.889805Z","shell.execute_reply":"2023-04-13T05:33:53.855367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Each vertical bar represents a feature. If there is no missing value, the bar will be as black as the rightmost one. Some areas are almost all white, which means that almost all the values of this feature are missing.","metadata":{}},{"cell_type":"markdown","source":"Drop meaningless values:","metadata":{}},{"cell_type":"code","source":"# Drop customer_ID and S_2 from train_df dataframe since they are not required for model building\ntrain_df.drop(axis=1, columns=['customer_ID','S_2'], inplace=True)\n# Drop S_2 in test_df dataframe which is not required for model building\ntest_df.drop(axis=1, columns=['S_2'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:33:53.858947Z","iopub.execute_input":"2023-04-13T05:33:53.859529Z","iopub.status.idle":"2023-04-13T05:33:54.012386Z","shell.execute_reply.started":"2023-04-13T05:33:53.859476Z","shell.execute_reply":"2023-04-13T05:33:54.010722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Handle categorial column:","metadata":{}},{"cell_type":"code","source":"label_encoder = LabelEncoder()\n\nfor feature in categorical_features:\n    train_df[feature] = label_encoder.fit_transform(train_df[feature])","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:33:54.014245Z","iopub.execute_input":"2023-04-13T05:33:54.014837Z","iopub.status.idle":"2023-04-13T05:33:54.478553Z","shell.execute_reply.started":"2023-04-13T05:33:54.014779Z","shell.execute_reply":"2023-04-13T05:33:54.476912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Drop columns leads to multicollinearity:","metadata":{}},{"cell_type":"code","source":"features_drop_for_multicollinearity = ['B_11', 'S_7', 'B_13', 'B_23', 'D_74', 'D_75', 'D_77', 'B_33', 'B_37', 'D_110', 'D_111']\ntrain_df.drop(columns=features_drop_for_multicollinearity)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:33:54.480709Z","iopub.execute_input":"2023-04-13T05:33:54.481072Z","iopub.status.idle":"2023-04-13T05:33:54.620671Z","shell.execute_reply.started":"2023-04-13T05:33:54.481037Z","shell.execute_reply":"2023-04-13T05:33:54.619278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Handle missing values:","metadata":{}},{"cell_type":"code","source":"# rough_train_data = train_df.copy()\nfill_mean_for_missing = [\n    'P_2','D_48','D_52','P_3','D_55','D_59','D_62',\n    'D_68','D_70','D_104',\n    'D_107','S_27','D_115','D_117','D_118',\n    'D_119','D_121','D_122','D_123','D_124',\n    'D_130'\n]\nfill_mode_for_missing = [\n    'S_3','D_44','D_46',\n    'B_13','D_61','D_69'\n    ,'D_78','D_79','D_83'\n    ,'D_84','D_89','D_91',\n    'D_102','R_27','D_113','D_114','D_116',\n    'D_120','D_125','D_128','D_129',\n    'D_131','D_133','D_139','D_140','D_141','D_143',\n    'D_144','D_145'\n]\ndrop_for_missing = [\n    'D_42','D_43','D_49',\n    'D_50','D_53','D_56'\n    ,'S_9','B_17','D_66'\n    ,'D_73','D_76','D_77'\n    ,'R_9','D_82','B_29'\n    ,'D_87','D_88','D_105'\n    ,'D_106','R_26','D_108'\n    ,'D_110','D_111','B_39'\n    ,'B_42','D_132','D_134'\n    ,'D_135','D_136','D_137'\n    ,'D_138','D_142'\n]\n# pd.options.display.max_info_columns = 300\n# rough_train_data.info()\nfor feature_name in fill_mean_for_missing:\n    train_df[feature_name].fillna(train_df[feature_name].mean(), inplace=True)\nfor feature_name in fill_mode_for_missing:\n    train_df[feature_name].fillna(train_df[feature_name].mode(), inplace=True)\nfor feature_name in drop_for_missing:\n    train_df.drop(axis=1, columns=[feature_name], inplace=True)\n# rough_train_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:36:34.591605Z","iopub.status.idle":"2023-04-13T05:36:34.592572Z","shell.execute_reply.started":"2023-04-13T05:36:34.592240Z","shell.execute_reply":"2023-04-13T05:36:34.592274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in train_df.columns.tolist():  \n    CpuKiller = train_df[column].mean()  \n    # Fill in each column\n    train_df[column] = train_df[column].fillna(CpuKiller) \npd.options.display.max_info_columns = 300","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:33:56.886293Z","iopub.execute_input":"2023-04-13T05:33:56.886779Z","iopub.status.idle":"2023-04-13T05:33:57.076231Z","shell.execute_reply.started":"2023-04-13T05:33:56.886727Z","shell.execute_reply":"2023-04-13T05:33:57.074901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualize the degree of dispersion of the data","metadata":{}},{"cell_type":"code","source":"train_df.boxplot()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:33:57.078163Z","iopub.execute_input":"2023-04-13T05:33:57.078585Z","iopub.status.idle":"2023-04-13T05:34:16.157608Z","shell.execute_reply.started":"2023-04-13T05:33:57.078548Z","shell.execute_reply":"2023-04-13T05:34:16.156265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Rank correlations\n#correlations = train_df.corr()['target'].abs().sort_values(ascending=False)\n\n# Select top 100 related features\n#top_100_features = correlations.index[1:51]\n\n# Create a new dataset\n#train_df_top_100 = train_df[top_100_features].copy()\n\n# Keep 'target' feature\n#train_df_top_100.loc[:, 'target'] = train_df['target']\n\n#train_df = train_df_top_100\n","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:34:16.159682Z","iopub.execute_input":"2023-04-13T05:34:16.160064Z","iopub.status.idle":"2023-04-13T05:34:16.166392Z","shell.execute_reply.started":"2023-04-13T05:34:16.160029Z","shell.execute_reply":"2023-04-13T05:34:16.164986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(columns='target')\ny = train_df['target']","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:34:16.168680Z","iopub.execute_input":"2023-04-13T05:34:16.169055Z","iopub.status.idle":"2023-04-13T05:34:16.251028Z","shell.execute_reply.started":"2023-04-13T05:34:16.169019Z","shell.execute_reply":"2023-04-13T05:34:16.249221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of X\", X.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:34:16.257133Z","iopub.execute_input":"2023-04-13T05:34:16.257585Z","iopub.status.idle":"2023-04-13T05:34:16.267650Z","shell.execute_reply.started":"2023-04-13T05:34:16.257546Z","shell.execute_reply":"2023-04-13T05:34:16.265953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of y\", y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:34:16.269243Z","iopub.execute_input":"2023-04-13T05:34:16.270047Z","iopub.status.idle":"2023-04-13T05:34:16.281057Z","shell.execute_reply.started":"2023-04-13T05:34:16.270003Z","shell.execute_reply":"2023-04-13T05:34:16.279225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, stratify=y, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T05:34:16.283397Z","iopub.execute_input":"2023-04-13T05:34:16.284318Z","iopub.status.idle":"2023-04-13T05:34:16.477381Z","shell.execute_reply.started":"2023-04-13T05:34:16.284263Z","shell.execute_reply":"2023-04-13T05:34:16.476026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import RobustScaler\n\n# Create pipeline with robust scaler and LGBM classifier\npipeline = make_pipeline(RobustScaler(), LGBMClassifier(random_state=45, n_estimators=5000))\n\n# Fit pipeline on training data\npipeline.fit(X_train, y_train)\n\n# Predict on test data\ny_pred = pipeline.predict(X_test)\n\n# Calculate the accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(\"Accuracy:\", accuracy)\n\n# Evaluate performance\nfrom sklearn.metrics import classification_report\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-04-13T07:07:23.783639Z","iopub.execute_input":"2023-04-13T07:07:23.785291Z","iopub.status.idle":"2023-04-13T07:12:17.070532Z","shell.execute_reply.started":"2023-04-13T07:07:23.785240Z","shell.execute_reply":"2023-04-13T07:12:17.069132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Parameter tuning:","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.model_selection import GridSearchCV\n\n# Define Parameters Distribution\n# Define Parameter grid\nparam_grid = {\n    'n_estimators': range(5000, 10000, 1000)\n}\n\nmodel = LGBMClassifier()\n\n# Cross validation and grid search\ngrid_search = GridSearchCV(model, param_grid, scoring='accuracy', cv=5, verbose=1)\ngrid_search.fit(X_train, y_train)\n\n# Print best parametors\nprint(\"Best parameters found:\", grid_search.best_params_)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T06:42:11.816677Z","iopub.execute_input":"2023-04-13T06:42:11.817385Z","iopub.status.idle":"2023-04-13T06:42:14.083325Z","shell.execute_reply.started":"2023-04-13T06:42:11.817329Z","shell.execute_reply":"2023-04-13T06:42:14.080664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Other data normalization methods for testing:","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\n\n# Create pipeline with standard scaler and LGBM classifier\npipeline = make_pipeline(StandardScaler(), LGBMClassifier(random_state=45, n_estimators=6000))\n\n# Fit pipeline on training data\npipeline.fit(X_train, y_train)\n\n# Predict on test data\ny_pred = pipeline.predict(X_test)\n\n# Calculate the accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(\"Accuracy:\", accuracy)\n\n# Evaluate performance\nfrom sklearn.metrics import classification_report\nprint(classification_report(y_test, y_pred))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import MinMaxScaler\n\n# Create pipeline with MinMaxScaler and LGBM classifier\npipeline = make_pipeline(MinMaxScaler(), LGBMClassifier(random_state=45, n_estimators=6000))\n\n# Fit pipeline on training data\npipeline.fit(X_train, y_train)\n\n# Predict on test data\ny_pred = pipeline.predict(X_test)\n\n# Calculate the accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(\"Accuracy:\", accuracy)\n\n# Evaluate performance\nfrom sklearn.metrics import classification_report\nprint(classification_report(y_test, y_pred))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import Normalizer\n\n# Create pipeline with Normalizer and LGBM classifier\npipeline = make_pipeline(Normalizer(), LGBMClassifier(random_state=45, n_estimators=6000))\n\n# Fit pipeline on training data\npipeline.fit(X_train, y_train)\n\n# Predict on test data\ny_pred = pipeline.predict(X_test)\n\n# Calculate the accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(\"Accuracy:\", accuracy)\n\n# Evaluate performance\nfrom sklearn.metrics import classification_report\nprint(classification_report(y_test, y_pred))","metadata":{},"execution_count":null,"outputs":[]}]}