{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🔥 TPS-JUN22 Unsupervised Clustering + GBDT\nHello Kaggle After a few days off from the competition.\n\nI saw multiple **Notebooks** using **Classification** models to enhance the predictions provided by the Clustering Algorithm; It took me a while to understand the technique due to majority of the **Notebooks** and **Code** don't have good documentation, in this **Notebbook** I will explain to my best how this type of technique works.\n\n**This will be the strategy will follow on this Notebook:**\n\n* Loading all the information available, and quickly explore it.\n* We identify the Clusters using a Clustering Algortihm (Bayesian Gaussian Mixture).\n* We create a sumset of the dataset, based on the Clusters with the highest probability to be that cluster (Greather than 80%).\n* We build a ML model passing the high probability records and the cluster number as the target.\n* We predict the cluster based on the new trained model predictions (GBTD or Other).\n\n\n**Credits and References:**\n* https://www.kaggle.com/code/adaubas/tps-jul22-lgbm-extratree-qda-soft-voting/notebook?scriptVersionId=100798978\n* https://www.kaggle.com/code/cv13j0/tps-jun22-unsupervised-clustering-with-keras\n* ...","metadata":{}},{"cell_type":"markdown","source":"# 1.0 Importing Libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-24T22:44:36.688531Z","iopub.execute_input":"2022-07-24T22:44:36.689527Z","iopub.status.idle":"2022-07-24T22:44:36.698630Z","shell.execute_reply.started":"2022-07-24T22:44:36.689482Z","shell.execute_reply":"2022-07-24T22:44:36.697764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Model Libraries for K-Means\nfrom sklearn.cluster import KMeans\nfrom sklearn import metrics\nfrom sklearn.decomposition import PCA\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import balanced_accuracy_score, roc_auc_score\nfrom sklearn.ensemble import ExtraTreesClassifier\n\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, PowerTransformer\n\n# Scientific Libraries for Calculations\nfrom scipy.spatial.distance import cdist\n\n# Visualization Libraries\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Import Machine Learning Model\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:36.700575Z","iopub.execute_input":"2022-07-24T22:44:36.701176Z","iopub.status.idle":"2022-07-24T22:44:36.714656Z","shell.execute_reply.started":"2022-07-24T22:44:36.701142Z","shell.execute_reply":"2022-07-24T22:44:36.713296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.0 Configuring the WorkSpace / Notebook","metadata":{}},{"cell_type":"code","source":"%%time\n# I like to disable my Notebook Warnings To Reduce Noice.\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:36.716523Z","iopub.execute_input":"2022-07-24T22:44:36.717193Z","iopub.status.idle":"2022-07-24T22:44:36.730838Z","shell.execute_reply.started":"2022-07-24T22:44:36.717159Z","shell.execute_reply":"2022-07-24T22:44:36.729892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Notebook Configuration...\n\n# Amount of data we want to load into the Model...\nDATA_ROWS = None\n# Dataframe, the amount of rows and cols to visualize...\nNROWS = 25\nNCOLS = 20\n# Main data location path...\nBASE_PATH = '...'","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:36.733053Z","iopub.execute_input":"2022-07-24T22:44:36.733685Z","iopub.status.idle":"2022-07-24T22:44:36.742558Z","shell.execute_reply.started":"2022-07-24T22:44:36.733641Z","shell.execute_reply":"2022-07-24T22:44:36.741576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Configure notebook display settings to only use 2 decimal places, tables look nicer.\npd.options.display.float_format = '{:,.2f}'.format\npd.set_option('display.max_columns', NCOLS) \npd.set_option('display.max_rows', NROWS)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:36.743861Z","iopub.execute_input":"2022-07-24T22:44:36.744714Z","iopub.status.idle":"2022-07-24T22:44:36.755748Z","shell.execute_reply.started":"2022-07-24T22:44:36.744671Z","shell.execute_reply":"2022-07-24T22:44:36.754496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3.0 Loading the Datasets, Pandas","metadata":{}},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\ndataset = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')\nsubmission = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:36.757141Z","iopub.execute_input":"2022-07-24T22:44:36.758189Z","iopub.status.idle":"2022-07-24T22:44:37.630365Z","shell.execute_reply.started":"2022-07-24T22:44:36.758151Z","shell.execute_reply":"2022-07-24T22:44:37.629202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4.0 Exploring the Loaded Dataset","metadata":{}},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\ndataset.info(verbose = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:37.631674Z","iopub.execute_input":"2022-07-24T22:44:37.632008Z","iopub.status.idle":"2022-07-24T22:44:37.817549Z","shell.execute_reply.started":"2022-07-24T22:44:37.631963Z","shell.execute_reply":"2022-07-24T22:44:37.816324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:37.820409Z","iopub.execute_input":"2022-07-24T22:44:37.820715Z","iopub.status.idle":"2022-07-24T22:44:37.843738Z","shell.execute_reply.started":"2022-07-24T22:44:37.820679Z","shell.execute_reply":"2022-07-24T22:44:37.842977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\ndataset.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:37.844836Z","iopub.execute_input":"2022-07-24T22:44:37.845419Z","iopub.status.idle":"2022-07-24T22:44:38.046447Z","shell.execute_reply.started":"2022-07-24T22:44:37.845388Z","shell.execute_reply":"2022-07-24T22:44:38.045429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\ndataset.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:38.048123Z","iopub.execute_input":"2022-07-24T22:44:38.048552Z","iopub.status.idle":"2022-07-24T22:44:38.178368Z","shell.execute_reply.started":"2022-07-24T22:44:38.048510Z","shell.execute_reply":"2022-07-24T22:44:38.177181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\ndataset.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:38.179925Z","iopub.execute_input":"2022-07-24T22:44:38.180349Z","iopub.status.idle":"2022-07-24T22:44:38.196563Z","shell.execute_reply.started":"2022-07-24T22:44:38.180308Z","shell.execute_reply":"2022-07-24T22:44:38.195459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\ndataset.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:38.198244Z","iopub.execute_input":"2022-07-24T22:44:38.198888Z","iopub.status.idle":"2022-07-24T22:44:38.217812Z","shell.execute_reply.started":"2022-07-24T22:44:38.198843Z","shell.execute_reply":"2022-07-24T22:44:38.217002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5.0 Selecting Model Features","metadata":{}},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\ncategorical_cols = ['f_07','f_08','f_09','f_10','f_11','f_12','f_13']","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:38.219128Z","iopub.execute_input":"2022-07-24T22:44:38.219453Z","iopub.status.idle":"2022-07-24T22:44:38.224947Z","shell.execute_reply.started":"2022-07-24T22:44:38.219427Z","shell.execute_reply":"2022-07-24T22:44:38.223963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Create a Feature List...\nignore = ['id']\nfeat = [feat for feat in dataset.columns if feat not in ignore]\n\n# Overwrite the feature by this default list...\nfeat =['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:38.226265Z","iopub.execute_input":"2022-07-24T22:44:38.226574Z","iopub.status.idle":"2022-07-24T22:44:38.236144Z","shell.execute_reply.started":"2022-07-24T22:44:38.226535Z","shell.execute_reply":"2022-07-24T22:44:38.235152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6.0 Pre-Processing the Data, Normalization","metadata":{}},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\nX = dataset.copy(deep = True)\nX[feat] = StandardScaler().fit(X[feat]).transform(X[feat])\nX[feat] = PowerTransformer().fit(X[feat]).transform(X[feat])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:38.237503Z","iopub.execute_input":"2022-07-24T22:44:38.237850Z","iopub.status.idle":"2022-07-24T22:44:40.356212Z","shell.execute_reply.started":"2022-07-24T22:44:38.237819Z","shell.execute_reply":"2022-07-24T22:44:40.355281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7.0 Building a BGM Model","metadata":{}},{"cell_type":"code","source":"%%time\n# Set a baseline with K-Means\n# 10 clusters\nn_components = 7\n# Runs in parallel All CPUs\n# Train Gaussian Mixture.\n\nbgm = BayesianGaussianMixture(n_components = n_components, \n                              covariance_type = 'full', \n                              tol = 0.001, \n                              reg_covar = 1e-06, \n                              max_iter = 100, \n                              n_init = 1, \n                              init_params = 'kmeans', \n                              weight_concentration_prior_type = 'dirichlet_process', \n                              weight_concentration_prior = None, \n                              mean_precision_prior = None, \n                              mean_prior = None, \n                              degrees_of_freedom_prior = None, \n                              covariance_prior = None, \n                              random_state = 1, \n                              warm_start = False, \n                              verbose = 0, \n                              verbose_interval = 10)\n\nbgm.fit(X[feat])\nbgm_predictions = bgm.predict(X[feat])\nbgm_predictions_proba = bgm.predict_proba(X[feat])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:44:40.357502Z","iopub.execute_input":"2022-07-24T22:44:40.357817Z","iopub.status.idle":"2022-07-24T22:45:11.764383Z","shell.execute_reply.started":"2022-07-24T22:44:40.357787Z","shell.execute_reply":"2022-07-24T22:45:11.763474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 8.0 Creating Predictions using the BGM Model","metadata":{}},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\nsubmission[\"Predicted\"] = bgm_predictions\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:45:11.766131Z","iopub.execute_input":"2022-07-24T22:45:11.766930Z","iopub.status.idle":"2022-07-24T22:45:11.887382Z","shell.execute_reply.started":"2022-07-24T22:45:11.766879Z","shell.execute_reply":"2022-07-24T22:45:11.886414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\nsubmission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:45:11.891112Z","iopub.execute_input":"2022-07-24T22:45:11.892314Z","iopub.status.idle":"2022-07-24T22:45:11.903397Z","shell.execute_reply.started":"2022-07-24T22:45:11.892265Z","shell.execute_reply":"2022-07-24T22:45:11.902007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 9.0 Building a Pseudo-Labeling Dataset to Train Next Model","metadata":{}},{"cell_type":"code","source":"%%time\n\n# Building a high confidence dataset for training...\nX['predictions'] = bgm_predictions\nX['predict_proba'] = 0\n\nCLUSTERS = 7 # Number of Clusters or Components used...\n\nfor n in range(CLUSTERS):\n    # Loop over all the clusters, and creates a probability column for each cluster or component...\n    X[f'predict_proba_{n}'] = bgm_predictions_proba[:,n] # Write the probability for each cluster as a new feature\n    X.loc[X.predictions == n,'predict_proba'] = X[f'predict_proba_{n}'] # Extract the probility of the estimated cluster.","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:45:11.907139Z","iopub.execute_input":"2022-07-24T22:45:11.907879Z","iopub.status.idle":"2022-07-24T22:45:11.942931Z","shell.execute_reply.started":"2022-07-24T22:45:11.907840Z","shell.execute_reply":"2022-07-24T22:45:11.942200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\npct_data = 0\nmin_probability = 0.75 # Set the minimun probability allowed to create the new training dataset...\n\n# Generate a list of indexes with higher confidence or probabilities\n\nhigh_confidence_idx = np.array([])\nfor cluster in range(CLUSTERS):\n    median_probability = X[X['predictions'] == cluster]['predict_proba'].median()\n    idx = X[(X['predictions'] == cluster) & (X['predict_proba'] > min_probability)].index\n    pct_data = len(idx) / len(X[(X['predictions'] == cluster)])\n    print(f'Cluster: {cluster}, Median Probability: {median_probability : .3f}, Data Pct: {pct_data: .2f}')                           \n    high_confidence_idx = np.concatenate((high_confidence_idx, idx))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:45:11.944013Z","iopub.execute_input":"2022-07-24T22:45:11.944664Z","iopub.status.idle":"2022-07-24T22:45:12.059444Z","shell.execute_reply.started":"2022-07-24T22:45:11.944631Z","shell.execute_reply":"2022-07-24T22:45:12.058362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 10.0 Developing a LGBM Model Using the Pseudo-Labeling","metadata":{}},{"cell_type":"code","source":"%%time\nN_FOLDS = 10\n\nX_trn = X.loc[high_confidence_idx][feat]\nlabel = X.loc[high_confidence_idx]['predictions']\n\n\nparams_lgb = {'learning_rate': 0.07,\n              'objective': 'multiclass',\n              'boosting': 'gbdt',\n              'verbosity': -1,\n              'n_jobs': -1,\n              'num_classes':CLUSTERS} \n\nmodel_list = []\ngkf = StratifiedKFold(N_FOLDS)\nlgbm_predictions_prob = 0\n\nfor fold, (train_idx, valid_idx) in enumerate(gkf.split(X_trn,label)):  \n    print(f'FOLD:{fold}...')\n    trn_dataset = lgb.Dataset(X_trn.iloc[train_idx],label.iloc[train_idx],feature_name = feat)\n    val_dataset = lgb.Dataset(X_trn.iloc[valid_idx],label.iloc[valid_idx],feature_name = feat)\n    \n    model = lgb.train(params = params_lgb, \n                      train_set = trn_dataset, \n                      valid_sets = val_dataset, \n                      num_boost_round = 5000, \n                      callbacks = [lgb.early_stopping(stopping_rounds = 300, verbose = True), lgb.log_evaluation(period = 200)])  \n    \n    model_list.append(model)\n    \n    y_pred_proba = model.predict(X_trn.iloc[valid_idx])\n    y_pred = np.argmax(y_pred_proba, axis = 1)\n    \n    score = balanced_accuracy_score(label.iloc[valid_idx], y_pred)\n    auc = roc_auc_score(label.iloc[valid_idx], y_pred_proba, average = \"weighted\", multi_class = \"ovo\")\n    \n    lgbm_predictions_prob += model.predict(X[feat]) / N_FOLDS\n    \n    print(f'LGBM AUC : {score:.3f} | ACC : {auc:.1%}\\n')\n    print('.' * 10)\n    print('')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:45:12.060893Z","iopub.execute_input":"2022-07-24T22:45:12.061210Z","iopub.status.idle":"2022-07-24T22:53:06.074449Z","shell.execute_reply.started":"2022-07-24T22:45:12.061181Z","shell.execute_reply":"2022-07-24T22:53:06.073413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 11.0 Developing a Extra Trees Model Using the Pseudo-Labeling","metadata":{}},{"cell_type":"code","source":"%%time\nN_FOLDS = 5\nSEED = 15\n\nX_trn = X.loc[high_confidence_idx][feat]\nlabel = X.loc[high_confidence_idx]['predictions']\n\nmodel_list = []\ngkf = StratifiedKFold(N_FOLDS)\nextratrees_predictions_prob = 0\n\nfor fold, (train_idx, valid_idx) in enumerate(gkf.split(X_trn,label)):  \n    print(f'FOLD:{fold}...')\n    X_train, y_train = X_trn.iloc[train_idx], label.iloc[train_idx]\n    X_valid, y_valid = X_trn.iloc[valid_idx], label.iloc[valid_idx]\n    \n    model = ExtraTreesClassifier(n_estimators = 100, random_state = SEED)\n    model.fit(X_train, y_train)\n    model_list.append(model)\n    \n    y_pred = model.predict(X_valid)\n    y_pred_proba = model.predict_proba(X_valid)\n    \n    score = balanced_accuracy_score(y_valid, y_pred)\n    auc = roc_auc_score(y_valid, y_pred_proba, average = \"weighted\", multi_class = \"ovo\")\n    \n    extratrees_predictions_prob += model.predict_proba(X[feat]) / N_FOLDS\n    \n    print(f'Extra Trees AUC : {score:.3f} | ACC : {auc:.1%}')\n    print('.' * 10)\n    print('')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:53:06.075900Z","iopub.execute_input":"2022-07-24T22:53:06.076271Z","iopub.status.idle":"2022-07-24T22:53:55.345275Z","shell.execute_reply.started":"2022-07-24T22:53:06.076218Z","shell.execute_reply":"2022-07-24T22:53:55.344058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# 12.0 Utilizing A Soft-Voting Strategy","metadata":{}},{"cell_type":"code","source":"%%time\ndef soft_voting_predictions(prediction_prob, df):\n    '''\n    \n    '''\n    values = list(range(CLUSTERS))\n    pred_test = pd.DataFrame(np.zeros((dataset.shape[0], 7)), columns = values)\n    \n    for model, probability in enumerate(prediction_prob):\n        max_value = np.argmax(probability, axis = 1)\n        df[f'pred_model_{model}'] = max_value\n        \n        # Sort the Predictions...\n        pred_keys = df[f'pred_model_{model}'].value_counts().index.to_list()\n        pred_dict = dict(zip(pred_keys, values))\n        df[f'pred_model_{model}'] = df[f'pred_model_{model}'].map(pred_dict)\n        \n        pred_new = pd.DataFrame(probability).rename(columns = pred_dict)   \n        pred_new = pred_new.reindex(sorted(pred_new.columns), axis=1)\n        \n        # Soft Voting by Probabiliy Addition\n        pred_test += pred_new \n                                      \n    return np.argmax(np.array(pred_test), axis = 1)                                 ","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:53:55.347100Z","iopub.execute_input":"2022-07-24T22:53:55.347782Z","iopub.status.idle":"2022-07-24T22:53:55.356525Z","shell.execute_reply.started":"2022-07-24T22:53:55.347726Z","shell.execute_reply":"2022-07-24T22:53:55.355802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsoft_vote_results = soft_voting_predictions([lgbm_predictions_prob, extratrees_predictions_prob], dataset)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:53:55.357760Z","iopub.execute_input":"2022-07-24T22:53:55.358247Z","iopub.status.idle":"2022-07-24T22:53:55.407002Z","shell.execute_reply.started":"2022-07-24T22:53:55.358217Z","shell.execute_reply":"2022-07-24T22:53:55.406220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsoft_vote_results","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:53:55.408273Z","iopub.execute_input":"2022-07-24T22:53:55.408737Z","iopub.status.idle":"2022-07-24T22:53:55.415376Z","shell.execute_reply.started":"2022-07-24T22:53:55.408706Z","shell.execute_reply":"2022-07-24T22:53:55.414450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 13.0 Creating a Submission File","metadata":{}},{"cell_type":"code","source":"%%time\nsubmission_lgbm = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsubmission_lgbm['Predicted'] = np.argmax(lgbm_predictions_prob, axis = 1)\nsubmission_lgbm.to_csv(\"submission_lgbm.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:53:55.416364Z","iopub.execute_input":"2022-07-24T22:53:55.416737Z","iopub.status.idle":"2022-07-24T22:53:55.540611Z","shell.execute_reply.started":"2022-07-24T22:53:55.416708Z","shell.execute_reply":"2022-07-24T22:53:55.539738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsubmission_soft_vote = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsubmission_soft_vote['Predicted'] = soft_vote_results\nsubmission_soft_vote.to_csv(\"submission_softvote.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:53:55.541710Z","iopub.execute_input":"2022-07-24T22:53:55.542826Z","iopub.status.idle":"2022-07-24T22:53:55.657711Z","shell.execute_reply.started":"2022-07-24T22:53:55.542779Z","shell.execute_reply":"2022-07-24T22:53:55.656434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\nsubmission_soft_vote.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:53:55.659245Z","iopub.execute_input":"2022-07-24T22:53:55.659665Z","iopub.status.idle":"2022-07-24T22:53:55.671070Z","shell.execute_reply.started":"2022-07-24T22:53:55.659623Z","shell.execute_reply":"2022-07-24T22:53:55.670133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}