{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3696790,"sourceType":"datasetVersion","datasetId":2211601}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Import packages\nimport numpy as np\nimport pandas as pd\nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:34:32.115967Z","iopub.execute_input":"2024-12-05T07:34:32.11638Z","iopub.status.idle":"2024-12-05T07:34:33.377746Z","shell.execute_reply.started":"2024-12-05T07:34:32.116343Z","shell.execute_reply":"2024-12-05T07:34:33.37627Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# loading data and preprocessing","metadata":{}},{"cell_type":"code","source":"#Loading the train data\ntrain_data = pd.read_parquet('../train_data.parquet')\n\n# Select the first 20,000 rows\ntrain_data = train_data.head(20000)\n\n#Explore train data\ntrain_data.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Check shape of train data\ntrain_data.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Check for number of unique customers\nlen(train_data.customer_ID.unique())\n\n## We have 458913 unique customers.","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)\n\n# Check for number of missing values\n# Filter columns starting with specific letters\nprefixes = ['D', 'S', 'P', 'B', 'R']\nsummary = {}\n\nfor prefix in prefixes:\n    # Get columns that start with the prefix\n    filter_cols = [col for col in train_data.columns if col.startswith(prefix)]\n    \n    # Calculate the number of columns with missing values\n    num_cols_with_missing = sum(train_data[filter_cols].isnull().sum() > 0)\n    \n    # Calculate the average number of missing values\n    avg_missing_values = train_data[filter_cols].isnull().sum().mean()\n    \n    # Store results in the summary dictionary\n    summary[prefix] = {\n        'columns_with_missing': num_cols_with_missing,\n        'average_missing_values': avg_missing_values\n    }\n\n# Display the summary\nfor prefix, stats in summary.items():\n    print(f\"Prefix '{prefix}':\")\n    print(f\"  Number of columns with missing values: {stats['columns_with_missing']}\")\n    print(f\"  Average missing values per column: {stats['average_missing_values']}\\n\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observation: Many columns have an very large number of missing valuess. There are 5531451 row, thus, if a columns has missing values exceding 90% of 5531451, they will be dropped.**","metadata":{}},{"cell_type":"code","source":"# Check for percentage of defaults to percentage of non-defaults\nPercentage=len(train_data[train_data['target']==1])*100/len(train_data[train_data['target']==0])\nPercentage\n#Only about 33.1% of the data are with defaults","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Removing columns with missing values > 90%.\r- \nLets remove columns if there are >90% of missing values.\r\nGiven that there are many columns with large number of missing values, it is impractical to go through every single one of them to determine whether it is useful. \r\nFurthermore, we do not have information on the feature (e.g. actual name of the feature) except the type of variable\r\nBrute force is thus a practical option to weed out columns with too many missing values.\r\nSince about 33.1% of the data are defaults(66.9% non-defaults), it is safe to say that columns with >90% missing data are not useful.","metadata":{}},{"cell_type":"code","source":"train=train_data.dropna(axis=1, thresh=int(0.90*len(train_data)))\n\n#Checking the shape of new train data\ntrain.shape\n## We are now left with 152 columns\n## Columns before 191, after 152. 39 columns where dropped.","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Handling multiple transactions\r\nThere are multiple transactions. Lets take only the latest transaction from each customer.\r\nLatest transaction may have missing values, we will perform forward fill for those missing values.\r\nWe perform forward fill as the last known value is likely to be brought forward to the next transaction.\r\nWe then do a backfill if the first row happens to be NA.","metadata":{}},{"cell_type":"code","source":"train=train.set_index(['customer_ID'])\ntrain=train.ffill().bfill()\ntrain=train.reset_index()\ntrain=train.groupby('customer_ID').tail(1)\ntrain=train.set_index(['customer_ID'])\n\n#Drop date column since it is no longer relevant\ntrain.drop(['S_2'],axis=1,inplace=True)\n#Check for number of rows\ntrain.shape\n## We now have 458913 rows, which corresponds to the number of unique customers.","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lets check again for missing values\n# Check for missing values\nmissing_summary = train.isnull().sum()\n\n# Count columns with missing values\nnum_cols_with_missing = (missing_summary > 0).sum()\n\nif num_cols_with_missing == 0:\n    print(\"There are no more missing values in the dataset.\")\nelse:\n    print(f\"There are {num_cols_with_missing} columns with missing values.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Encoding non-numeric categorical values","metadata":{}},{"cell_type":"code","source":"#Identify columns which are not numeric\ntrain.select_dtypes(['object'])\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"D_63 and D_64 turns out to be categorical but are strings","metadata":{}},{"cell_type":"code","source":"#Perform one-hot encoding for D_63 and D_64\n#Drop columns D_63 and D_64 subsequently\ntrain_D63 = pd.get_dummies(train[['D_63']])\ntrain = pd.concat([train, train_D63], axis=1)\ntrain = train.drop(['D_63'], axis=1)\n\ntrain_D64 = pd.get_dummies(train[['D_64']])\ntrain = pd.concat([train, train_D64], axis=1)\ntrain = train.drop(['D_64'], axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Lets check for the columns\ntrain.columns\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We now have 158 columns including target but it would still be useful to reduce the dimensionality of the data","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Removing highly correlated features\r\nWe will drop out columns with absolute correlation of more than 90%","metadata":{}},{"cell_type":"code","source":"# We shall remove highly correlated features\ntrain_without_target=train.drop(['target'],axis=1)\ncor_matrix = train_without_target.corr().abs()\nupper_tri = cor_matrix.where(np.triu(np.ones(cor_matrix.shape),k=1).astype(np.bool))\n\nto_drop = [column for column in upper_tri.columns if any(upper_tri[column] > 0.90)]\ntrain_drop_highcorr=train.drop(to_drop,axis=1)\ntrain_drop_highcorr.shape\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Compute correlation matrix (absolute values for comparison)\ncor_matrix = train_without_target.corr().abs()\n\n# Extract the upper triangle of the correlation matrix\nupper_tri = cor_matrix.where(np.triu(np.ones(cor_matrix.shape), k=1).astype(bool))\n\n# Find the 10 most correlated pairs\nmost_corr_pairs = (\n    upper_tri.unstack()\n    .sort_values(ascending=False)\n    .dropna()\n    .head(10)\n)\n\n# Print the pairs and their correlation values\nprint(\"Top 10 Most Correlated Feature Pairs:\")\nfor pair, corr in most_corr_pairs.items():\n    print(f\"{pair[0]} and {pair[1]}: {corr:.2f}\")\n\n# Get the feature names for the top 10 pairs\nmost_corr_features = set(\n    [item for sublist in most_corr_pairs.index.to_list() for item in sublist]\n)\n\n# Filter the correlation matrix for those features\nfiltered_corr_matrix = cor_matrix.loc[most_corr_features, most_corr_features]\n\n# Plot the heatmap\nplt.figure(figsize=(10, 8))\nsns.heatmap(filtered_corr_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\")\nplt.title(\"Top 10 Most Correlated Features\")\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We are now left with 145 columns, which is easier to handle than 158","metadata":{}},{"cell_type":"markdown","source":"# Removing columns with low variance\r\nLets remove columns with low variance=0.1. Keep only columns with high variance","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_selection import VarianceThreshold\nfrom itertools import compress\ndef fs_variance(df, threshold:float=0.1):\n    \"\"\"\n    Return a list of selected variables based on the threshold.\n    \"\"\"\n    # The list of columns in the data frame\n    features = list(df.columns)\n    \n    # Initialize and fit the method\n    vt = VarianceThreshold(threshold = threshold)\n    _ = vt.fit(df)\n    \n    # Get which column names which pass the threshold\n    feat_select = list(compress(features, vt.get_support()))\n    \n    return feat_select\n    \ncolumns_to_keep=fs_variance(train_drop_highcorr)\n\n\ntrain_final=train[columns_to_keep]\nlen(columns_to_keep)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We are left with 54 columns (excluding target, 55th column), which passed the threshold.","metadata":{}},{"cell_type":"markdown","source":"# splitting the data and testing models","metadata":{}},{"cell_type":"code","source":"#Split the target into y. Remove target from x.\ny_train=train_final['target']\nx_train=train_final.drop(['target'],axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split train data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score  \nfrom sklearn.metrics import precision_score                         \nfrom sklearn.metrics import recall_score\nx_train_split, x_test_split, y_train_split, y_test_split = train_test_split(x_train, y_train, test_size=0.25, random_state=26)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score\n\n# Define the model\nlog_reg = LogisticRegression()\n\n# Define the hyperparameters to tune\nparam_grid = {\n    'C': [0.01, 0.1, 1, 10, 100],  # Regularization strength\n    'solver': ['liblinear', 'saga'],  # Solvers that support both l1 and l2 penalties\n    'penalty': ['l1', 'l2'],         # Type of regularization\n}\n\n# Use GridSearchCV to find the best hyperparameters\ngrid_search = GridSearchCV(log_reg, param_grid, scoring='accuracy', cv=5, verbose=1)\ngrid_search.fit(x_train_split, y_train_split)\n\n# Best hyperparameters\nprint(\"Best Hyperparameters:\", grid_search.best_params_)\n\n# Train the model with the best hyperparameters\nbest_model = grid_search.best_estimator_\n\n# Predictions and evaluation\ny_predict = best_model.predict(x_test_split)\n\nprint('\\nLogistics Regression Accuracy: {:.3f}'.format(accuracy_score(y_test_split, y_predict)))\nprint('\\nLogistics Regression Precision: {:.3f}'.format(precision_score(y_test_split, y_predict)))\nprint('\\nLogistics Regression Recall: {:.3f}'.format(recall_score(y_test_split, y_predict)))\n\n\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\n# Train the model\nrf_model = RandomForestClassifier(random_state=42)\nrf_model.fit(x_train_split, y_train_split)\n\n# Make predictions\nrf_predict = rf_model.predict(x_test_split)\n\n# Evaluate the model\nprint('\\nRandom Forest Accuracy: {:.3f}'.format(accuracy_score(y_test_split, rf_predict)))\nprint('\\nRandom Forest Precision: {:.3f}'.format(precision_score(y_test_split, rf_predict)))\nprint('\\nRandom Forest Recall: {:.3f}'.format(recall_score(y_test_split, rf_predict)))\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score\n\n# Define the SVM model\nsvm = SVC(random_state=42)\n\n# Define the hyperparameters to tune\nparam_grid = {\n    'C': [0.1, 1, 10, 100],          # Regularization parameter\n    'kernel': ['linear', 'rbf', 'poly'],  # Kernel types\n    'gamma': ['scale', 'auto'],      # Kernel coefficient\n}\n\n# Use GridSearchCV to find the best hyperparameters\ngrid_search = GridSearchCV(svm, param_grid, scoring='accuracy', cv=5, verbose=1)\ngrid_search.fit(x_train_split, y_train_split)\n\n# Best hyperparameters\nprint(\"Best Hyperparameters:\", grid_search.best_params_)\n\n# Train the model with the best hyperparameters\nbest_svm_model = grid_search.best_estimator_\n\n# Make predictions\nsvm_predict = best_svm_model.predict(x_test_split)\n\n# Evaluate the model\nprint('\\nSVM Accuracy: {:.3f}'.format(accuracy_score(y_test_split, svm_predict)))\nprint('\\nSVM Precision: {:.3f}'.format(precision_score(y_test_split, svm_predict)))\nprint('\\nSVM Recall: {:.3f}'.format(recall_score(y_test_split, svm_predict)))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n# Define the XGBoost model\nxgb = XGBClassifier(random_state=42, use_label_encoder=False, eval_metric='logloss')\n\n# Define the hyperparameters to tune\nparam_grid = {\n    'n_estimators': [50, 100, 200],           # Number of boosting rounds\n    'learning_rate': [0.01, 0.1, 0.2],       # Step size shrinkage\n    'max_depth': [3, 5, 7],                  # Maximum depth of a tree\n}\n\n# Use GridSearchCV to find the best hyperparameters\ngrid_search = GridSearchCV(xgb, param_grid, scoring='accuracy', cv=5, verbose=1)\ngrid_search.fit(x_train_split, y_train_split)\n\n# Best hyperparameters\nprint(\"Best Hyperparameters:\", grid_search.best_params_)\n\n# Train the model with the best hyperparameters\nbest_xgb_model = grid_search.best_estimator_\n\n# Make predictions\nxgb_predict = best_xgb_model.predict(x_test_split)\n\n# Evaluate the model\nprint('\\nXGBoost Accuracy: {:.3f}'.format(accuracy_score(y_test_split, xgb_predict)))\nprint('XGBoost Precision: {:.3f}'.format(precision_score(y_test_split, xgb_predict)))\nprint('XGBoost Recall: {:.3f}'.format(recall_score(y_test_split, xgb_predict)))\nprint('XGBoost F1 Score: {:.3f}'.format(f1_score(y_test_split, xgb_predict)))\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use Random Forest\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV, GridSearchCV\n# Number of trees in random forest\nn_estimators = [int(x) for x in np.linspace(start = 200, stop = 2000, num = 10)]\n# Number of features to consider at every split\nmax_features = ['auto', 'sqrt']\n# Maximum number of levels in tree\nmax_depth = [int(x) for x in np.linspace(10, 110, num = 11)]\nmax_depth.append(None)\n# Minimum number of samples required to split a node\nmin_samples_split = [2, 5, 10]\n# Minimum number of samples required at each leaf node\nmin_samples_leaf = [1, 2, 4]\n# Method of selecting samples for training each tree\nbootstrap = [True, False]\n# Create the random grid\nrandom_grid = {'n_estimators': n_estimators,\n               'max_features': max_features,\n               'max_depth': max_depth,\n               'min_samples_split': min_samples_split,\n               'min_samples_leaf': min_samples_leaf,\n               'bootstrap': bootstrap}\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nrandom_grid = {\n    'n_estimators': [100, 200, 300, 400],  # Limit to a smaller range\n    'max_features': ['auto', 'sqrt'],  # Use fewer options\n    'max_depth': [10, 20, 30, None],  # Focus on practical depths\n    'min_samples_split': [2, 5, 10],\n    'min_samples_leaf': [1, 2, 4],\n    'bootstrap': [True, False]\n}\n\n# Initialize RandomForestClassifier\nrf = RandomForestClassifier(random_state=42)\n\n# Use RandomizedSearchCV for faster tuning\nrf_random = RandomizedSearchCV(\n    estimator=rf,\n    param_distributions=random_grid,\n    n_iter=25,  # Fewer iterations for speed\n    cv=3,  # Reduce cross-validation folds\n    verbose=1,  # Moderate verbosity\n    random_state=42,\n    n_jobs=-1  # Use all available cores\n)\n\n# Fit the random search model\nrf_random.fit(x_train_split, y_train_split)\n\n# Print the best parameters and score\nprint(\"Best Parameters:\", rf_random.best_params_)\nprint(\"Best Score:\", rf_random.best_score_)\n\nbest_rf = rf_random.best_estimator_\ntest_accuracy = best_rf.score(x_test_split, y_test_split)\n\nprint(\"Test Accuracy with Best Parameters:\", test_accuracy)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = RandomForestClassifier(n_estimators=400, max_features='sqrt', bootstrap=True, max_depth=30, min_samples_leaf=1, min_samples_split=5, n_jobs=-1)\nrf_random = GridSearchCV(estimator = rf, param_grid = random_grid, cv = 3, verbose=1, n_jobs = -1)\n# Fit the random search model\nmodel.fit(x_train_split,y_train_split)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef evaluate(model, test_features, test_labels):\n    predictions = model.predict(test_features)\n    errors = abs(predictions - test_labels)\n    mape = 100 * np.mean(errors / test_labels)\n    accuracy = 100 - mape\n    print('Model Performance')\n    print('Average Error: {:0.4f} degrees.'.format(np.mean(errors)))\n    print('Accuracy = {:0.2f}%.'.format(accuracy))\n    \n    return accuracy\nbase_model = RandomForestClassifier(n_estimators = 10, random_state = 42)\nbase_model.fit(x_train_split, y_train_split)\nbase_accuracy = evaluate(base_model, x_test_split, y_test_split)\n\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_accuracy = evaluate(model, x_test_split, y_test_split)\n\nprint('Improvement of {:0.2f}%.'.format( 100 * (random_accuracy - base_accuracy) / base_accuracy))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del x_train_split, y_train_split\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make a list of columns that we want to load for test data. Remove one-hot encoded names and target (since these columns not in the test data)\ncolumns_to_keep1=list(columns_to_keep)\ncolumns_to_keep1=columns_to_keep1+['D_63','D_64','customer_ID']\ncolumns_to_keep1.remove('D_63_CO')\ncolumns_to_keep1.remove('D_63_CR')\ncolumns_to_keep1.remove('D_64_O')\ncolumns_to_keep1.remove('D_64_R')\ncolumns_to_keep1.remove('D_64_U')\ncolumns_to_keep1.remove('target')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Read in the test_data\ntest_data = pd.read_parquet('../input/amex-parquet/test_data.parquet',columns=columns_to_keep1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# There are multiple transactions. Lets take only the latest transaction from each customer.\n# Latest transaction may have missing values, we will perform forward fill for those missing values.\n# We do a backfill if the first row happens to be Na\ntest=test_data.set_index(['customer_ID'])\ntest=test.ffill().bfill()\ntest=test.reset_index()\ntest=test.groupby('customer_ID').tail(1)\ntest=test.set_index(['customer_ID'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Perform one-hot encoding for D_63 and D_64\n#Drop columns D_63 and D_64 subsequently\ntest_D63 = pd.get_dummies(test[['D_63']])\ntest = pd.concat([test, test_D63], axis=1)\ntest = test.drop(['D_63'], axis=1)\n\ntest_D64 = pd.get_dummies(test[['D_64']])\ntest = pd.concat([test, test_D64], axis=1)\ntest = test.drop(['D_64'], axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Keep columns that we want.\ncolumns_to_keep.remove('target')\ntest_final=test[columns_to_keep]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Predict probabilities of default\ny_test_predict=model.predict_proba(test_final)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Retrieve the probability of default\ny_predict_final=y_test_predict[:,1]\n\n#Reset index of test\ntest=test.reset_index()\n\n# Merge the prediction and customer_ID into submission dataframe\nsubmission = pd.DataFrame({\"customer_ID\":test.customer_ID,\"prediction\":y_predict_final})\n\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"success\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}