{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# TPS-JUN22 Iterative Clustering, Using Pandas Chain\n## In this Notebook I implement the learning from this PyData Session 🤯\n\nEffective Pandas I Matt Harrison I PyData Salt Lake City Meetup\n\nhttps://www.youtube.com/watch?v=zgbUk90aQ6A\n\nIn particular I focus the efforts on the utilization of Pandas chaining ability + pipe method.","metadata":{}},{"cell_type":"code","source":"%%time\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T18:21:47.869556Z","iopub.execute_input":"2022-07-30T18:21:47.869972Z","iopub.status.idle":"2022-07-30T18:21:47.881201Z","shell.execute_reply.started":"2022-07-30T18:21:47.869939Z","shell.execute_reply":"2022-07-30T18:21:47.880218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# I like to disable my Notebook Warnings To Reduce Noice.\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:47.958454Z","iopub.execute_input":"2022-07-30T18:21:47.959597Z","iopub.status.idle":"2022-07-30T18:21:47.965388Z","shell.execute_reply.started":"2022-07-30T18:21:47.959554Z","shell.execute_reply":"2022-07-30T18:21:47.964276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Notebook Configuration...\n\n# Amount of data we want to load into the Model...\nDATA_ROWS = None\n# Dataframe, the amount of rows and cols to visualize...\nNROWS = 100\nNCOLS = 15\n# Main data location path...\nBASE_PATH = '...'","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:48.025731Z","iopub.execute_input":"2022-07-30T18:21:48.027003Z","iopub.status.idle":"2022-07-30T18:21:48.033975Z","shell.execute_reply.started":"2022-07-30T18:21:48.026960Z","shell.execute_reply":"2022-07-30T18:21:48.032562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Configure notebook display settings to only use 2 decimal places, tables look nicer.\npd.options.display.float_format = '{:,.2f}'.format\npd.set_option('display.max_columns', NCOLS) \npd.set_option('display.max_rows', NROWS)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:48.124483Z","iopub.execute_input":"2022-07-30T18:21:48.125215Z","iopub.status.idle":"2022-07-30T18:21:48.132954Z","shell.execute_reply.started":"2022-07-30T18:21:48.125168Z","shell.execute_reply":"2022-07-30T18:21:48.131749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')\nsubmission = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:48.206861Z","iopub.execute_input":"2022-07-30T18:21:48.208304Z","iopub.status.idle":"2022-07-30T18:21:49.477260Z","shell.execute_reply.started":"2022-07-30T18:21:48.208229Z","shell.execute_reply":"2022-07-30T18:21:49.476050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.479507Z","iopub.execute_input":"2022-07-30T18:21:49.479973Z","iopub.status.idle":"2022-07-30T18:21:49.492269Z","shell.execute_reply.started":"2022-07-30T18:21:49.479930Z","shell.execute_reply":"2022-07-30T18:21:49.491175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncols = ['id', 'f_00', 'f_01', 'f_02', 'f_03', 'f_04', 'f_05', 'f_06', 'f_07',\n       'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_14', 'f_15', 'f_16',\n       'f_17', 'f_18', 'f_19', 'f_20', 'f_21', 'f_22', 'f_23', 'f_24', 'f_25',\n       'f_26', 'f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.493709Z","iopub.execute_input":"2022-07-30T18:21:49.494059Z","iopub.status.idle":"2022-07-30T18:21:49.503933Z","shell.execute_reply.started":"2022-07-30T18:21:49.494028Z","shell.execute_reply":"2022-07-30T18:21:49.502704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset[cols].dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.506952Z","iopub.execute_input":"2022-07-30T18:21:49.507461Z","iopub.status.idle":"2022-07-30T18:21:49.534185Z","shell.execute_reply.started":"2022-07-30T18:21:49.507417Z","shell.execute_reply":"2022-07-30T18:21:49.533075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset[cols].memory_usage(deep=True).sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.536827Z","iopub.execute_input":"2022-07-30T18:21:49.537222Z","iopub.status.idle":"2022-07-30T18:21:49.558921Z","shell.execute_reply.started":"2022-07-30T18:21:49.537191Z","shell.execute_reply":"2022-07-30T18:21:49.557446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.560704Z","iopub.execute_input":"2022-07-30T18:21:49.561893Z","iopub.status.idle":"2022-07-30T18:21:49.588507Z","shell.execute_reply.started":"2022-07-30T18:21:49.561846Z","shell.execute_reply":"2022-07-30T18:21:49.587660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.589929Z","iopub.execute_input":"2022-07-30T18:21:49.590670Z","iopub.status.idle":"2022-07-30T18:21:49.618444Z","shell.execute_reply.started":"2022-07-30T18:21:49.590626Z","shell.execute_reply":"2022-07-30T18:21:49.617348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.620054Z","iopub.execute_input":"2022-07-30T18:21:49.620813Z","iopub.status.idle":"2022-07-30T18:21:49.835401Z","shell.execute_reply.started":"2022-07-30T18:21:49.620768Z","shell.execute_reply":"2022-07-30T18:21:49.834054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.836812Z","iopub.execute_input":"2022-07-30T18:21:49.837183Z","iopub.status.idle":"2022-07-30T18:21:49.973552Z","shell.execute_reply.started":"2022-07-30T18:21:49.837150Z","shell.execute_reply":"2022-07-30T18:21:49.972239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n(dataset.\n select_dtypes('int').\n describe()\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:49.977608Z","iopub.execute_input":"2022-07-30T18:21:49.978617Z","iopub.status.idle":"2022-07-30T18:21:50.040659Z","shell.execute_reply.started":"2022-07-30T18:21:49.978580Z","shell.execute_reply":"2022-07-30T18:21:50.039468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n(dataset.\n select_dtypes('float').\n describe()\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:50.042207Z","iopub.execute_input":"2022-07-30T18:21:50.042646Z","iopub.status.idle":"2022-07-30T18:21:50.211440Z","shell.execute_reply.started":"2022-07-30T18:21:50.042604Z","shell.execute_reply":"2022-07-30T18:21:50.210343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnp.iinfo(np.int8)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:50.212752Z","iopub.execute_input":"2022-07-30T18:21:50.213107Z","iopub.status.idle":"2022-07-30T18:21:50.221468Z","shell.execute_reply.started":"2022-07-30T18:21:50.213075Z","shell.execute_reply":"2022-07-30T18:21:50.220402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnp.iinfo(np.int16)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:50.222728Z","iopub.execute_input":"2022-07-30T18:21:50.223095Z","iopub.status.idle":"2022-07-30T18:21:50.234657Z","shell.execute_reply.started":"2022-07-30T18:21:50.223052Z","shell.execute_reply":"2022-07-30T18:21:50.233562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnp.finfo(np.float32)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:50.235676Z","iopub.execute_input":"2022-07-30T18:21:50.236083Z","iopub.status.idle":"2022-07-30T18:21:50.249264Z","shell.execute_reply.started":"2022-07-30T18:21:50.236050Z","shell.execute_reply":"2022-07-30T18:21:50.247994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# chaining \n(dataset\n [cols]\n .astype({'f_07': 'int8','f_08': 'int8','f_09': 'int8','f_10': 'int8','f_11': 'int8','f_12': 'int8','f_13': 'int8'})\n .select_dtypes('int8')\n .describe()\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:50.252384Z","iopub.execute_input":"2022-07-30T18:21:50.252722Z","iopub.status.idle":"2022-07-30T18:21:50.336141Z","shell.execute_reply.started":"2022-07-30T18:21:50.252690Z","shell.execute_reply":"2022-07-30T18:21:50.334980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# chaining \nfrom sklearn.preprocessing import StandardScaler, PowerTransformer\n\ndef stand_scaler(df, cols):\n    scaler = StandardScaler()\n    df[cols] = scaler.fit_transform(df[cols])\n    return df\n\ndef power_scaler(df, cols):\n    scaler = PowerTransformer()\n    df[cols] = scaler.fit_transform(df[cols])\n    return df\n\n    \ndef improve_memory(df):\n    return (dataset\n            [cols]\n            #.assign(f_sum = dataset.sum(axis = 1),\n            #        f_min = dataset.min(axis = 1),\n            #        f_max = dataset.max(axis = 1),\n            #        f_std = dataset.std(axis = 1),\n            #        f_mad = dataset.mad(axis = 1),\n            #        f_avg = dataset.mean(axis = 1),\n            #       )\n            .astype({'f_07': 'int8','f_08': 'int8','f_09': 'int8','f_10': 'int8','f_11': 'int8','f_12': 'int8','f_13': 'int8'})\n            .pipe(stand_scaler, cols) # apply stand_scaler\n            .pipe(power_scaler, cols) # apply power_scaler\n            .select_dtypes(['int8', 'float64'])\n            .drop('id', axis = 1) # drop non-requiered columns\n            #.describe()\n           )\n\ntrain_dataset = improve_memory(dataset)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:50.337972Z","iopub.execute_input":"2022-07-30T18:21:50.338675Z","iopub.status.idle":"2022-07-30T18:21:54.996594Z","shell.execute_reply.started":"2022-07-30T18:21:50.338629Z","shell.execute_reply":"2022-07-30T18:21:54.995528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:54.998153Z","iopub.execute_input":"2022-07-30T18:21:54.999339Z","iopub.status.idle":"2022-07-30T18:21:55.006252Z","shell.execute_reply.started":"2022-07-30T18:21:54.999289Z","shell.execute_reply":"2022-07-30T18:21:55.005252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\n\n# Set a baseline with BayesianGaussianMixture\nn_components = 7\n# Runs in parallel All CPUs\n# Train Gaussian Mixture.\n\nbgm = BayesianGaussianMixture(n_components = n_components, \n                              covariance_type = 'full', \n                              tol = 0.001, \n                              reg_covar = 1e-06, \n                              max_iter = 256, \n                              n_init = 3, \n                              init_params = 'kmeans', \n                              weight_concentration_prior_type = 'dirichlet_process', \n                              weight_concentration_prior = None, \n                              mean_precision_prior = None, \n                              mean_prior = None, \n                              degrees_of_freedom_prior = None, \n                              covariance_prior = None, \n                              random_state = 1, \n                              warm_start = False, \n                              verbose = 0, \n                              verbose_interval = 10)\n\nbgm.fit(train_dataset)\nbgm_predictions = bgm.predict(train_dataset)\nbgm_predictions_proba = bgm.predict_proba(train_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:21:55.008079Z","iopub.execute_input":"2022-07-30T18:21:55.008527Z","iopub.status.idle":"2022-07-30T18:24:50.933998Z","shell.execute_reply.started":"2022-07-30T18:21:55.008494Z","shell.execute_reply":"2022-07-30T18:24:50.932707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\nsubmission[\"Predicted\"] = bgm_predictions\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:24:50.938851Z","iopub.execute_input":"2022-07-30T18:24:50.939681Z","iopub.status.idle":"2022-07-30T18:24:51.127734Z","shell.execute_reply.started":"2022-07-30T18:24:50.939629Z","shell.execute_reply":"2022-07-30T18:24:51.126835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\nsubmission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:24:51.129286Z","iopub.execute_input":"2022-07-30T18:24:51.129931Z","iopub.status.idle":"2022-07-30T18:24:51.142880Z","shell.execute_reply.started":"2022-07-30T18:24:51.129873Z","shell.execute_reply":"2022-07-30T18:24:51.141604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Building a high confidence dataset for training...\ntrain_dataset['predictions'] = bgm_predictions\ntrain_dataset['predict_proba'] = 0\n\nCLUSTERS = 7 # Number of Clusters or Components used...\n\nfor n in range(CLUSTERS):\n    # Loop over all the clusters, and creates a probability column for each cluster or component...\n    train_dataset[f'predict_proba_{n}'] = bgm_predictions_proba[:,n] # Write the probability for each cluster as a new feature\n    train_dataset.loc[train_dataset.predictions == n,'predict_proba'] = train_dataset[f'predict_proba_{n}'] # Extract the probility of the estimated cluster.","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:24:51.146453Z","iopub.execute_input":"2022-07-30T18:24:51.146796Z","iopub.status.idle":"2022-07-30T18:24:51.182290Z","shell.execute_reply.started":"2022-07-30T18:24:51.146759Z","shell.execute_reply":"2022-07-30T18:24:51.181394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\npct_data = 0\nmin_probability = 0.775 # Set the minimun probability allowed to create the new training dataset...\n\n# Generate a list of indexes with higher confidence or probabilities\n\nhigh_confidence_idx = np.array([])\nfor cluster in range(CLUSTERS):\n    median_probability = train_dataset[train_dataset['predictions'] == cluster]['predict_proba'].median()\n    idx = train_dataset[(train_dataset['predictions'] == cluster) & (train_dataset['predict_proba'] > min_probability)].index\n    pct_data = len(idx) / len(train_dataset[(train_dataset['predictions'] == cluster)])\n    print(f'Cluster: {cluster}, Median Probability: {median_probability : .3f}, Data Pct: {pct_data: .2f}')                           \n    high_confidence_idx = np.concatenate((high_confidence_idx, idx))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:24:51.183571Z","iopub.execute_input":"2022-07-30T18:24:51.184101Z","iopub.status.idle":"2022-07-30T18:24:51.293157Z","shell.execute_reply.started":"2022-07-30T18:24:51.184069Z","shell.execute_reply":"2022-07-30T18:24:51.291874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.columns\n\nfeat = ['f_00', 'f_01', 'f_02', 'f_03', 'f_04', 'f_05', 'f_06', 'f_07', 'f_08',\n       'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_14', 'f_15', 'f_16', 'f_17',\n       'f_18', 'f_19', 'f_20', 'f_21', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26',\n       'f_27', 'f_28',]\n\n# Optimal features with more signal...\nfeat = ['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:24:51.294579Z","iopub.execute_input":"2022-07-30T18:24:51.294965Z","iopub.status.idle":"2022-07-30T18:24:51.302442Z","shell.execute_reply.started":"2022-07-30T18:24:51.294926Z","shell.execute_reply":"2022-07-30T18:24:51.301233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import balanced_accuracy_score, roc_auc_score\n\nimport lightgbm as lgb \n\nN_FOLDS = 10\nCLUSTERS = 7\n\nX_trn = train_dataset.loc[high_confidence_idx][feat]\nlabel = train_dataset.loc[high_confidence_idx]['predictions']\n\n\nparams_lgb = {'learning_rate': 0.07,\n              'objective': 'multiclass',\n              'boosting': 'gbdt',\n              'verbosity': -1,\n              'n_jobs': -1,\n              'num_classes':CLUSTERS} \n\nmodel_list = []\ngkf = StratifiedKFold(N_FOLDS)\nlgbm_predictions_prob = 0\n\nfor fold, (train_idx, valid_idx) in enumerate(gkf.split(X_trn,label)):  \n    print(f'FOLD:{fold}...')\n    trn_dataset = lgb.Dataset(X_trn.iloc[train_idx],label.iloc[train_idx], feature_name = feat)\n    val_dataset = lgb.Dataset(X_trn.iloc[valid_idx],label.iloc[valid_idx], feature_name = feat)\n    \n    model = lgb.train(params = params_lgb, \n                      train_set = trn_dataset, \n                      valid_sets = val_dataset, \n                      num_boost_round = 10_000, \n                      callbacks = [lgb.early_stopping(stopping_rounds = 300, verbose = True), lgb.log_evaluation(period = 200)])  \n    \n    model_list.append(model)\n    \n    y_pred_proba = model.predict(X_trn.iloc[valid_idx])\n    y_pred = np.argmax(y_pred_proba, axis = 1)\n    \n    score = balanced_accuracy_score(label.iloc[valid_idx], y_pred)\n    auc = roc_auc_score(label.iloc[valid_idx], y_pred_proba, average = \"weighted\", multi_class = \"ovo\")\n    \n    lgbm_predictions_prob += model.predict(train_dataset[feat]) / N_FOLDS\n    \n    print(f'LGBM AUC : {score:.3f} | ACC : {auc:.1%}\\n')\n    print('.' * 10)\n    print('')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:24:51.303878Z","iopub.execute_input":"2022-07-30T18:24:51.304322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_dataset['lgbm_pred'] = np.argmax(lgbm_predictions_prob, axis = 1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_feat = ['f_00', 'f_01', 'f_02', 'f_03', 'f_04', 'f_05', 'f_06', 'f_07', 'f_08',\n       'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_14', 'f_15', 'f_16', 'f_17',\n       'f_18', 'f_19', 'f_20', 'f_21', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26',\n       'f_27', 'f_28', 'lgbm_pred']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\n\n# Set a baseline with BayesianGaussianMixture\nn_components = 7\n# Runs in parallel All CPUs\n# Train Gaussian Mixture.\n\nbgm = BayesianGaussianMixture(n_components = n_components, \n                              covariance_type = 'full', \n                              tol = 0.001, \n                              reg_covar = 1e-06, \n                              max_iter = 256, \n                              n_init = 3, \n                              init_params = 'kmeans', \n                              weight_concentration_prior_type = 'dirichlet_process', \n                              weight_concentration_prior = None, \n                              mean_precision_prior = None, \n                              mean_prior = None, \n                              degrees_of_freedom_prior = None, \n                              covariance_prior = None, \n                              random_state = 1, \n                              warm_start = False, \n                              verbose = 0, \n                              verbose_interval = 10)\n\nbgm.fit(train_dataset[selected_feat])\nbgm_predictions = bgm.predict(train_dataset[selected_feat])\nbgm_predictions_proba = bgm.predict_proba(train_dataset[selected_feat])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Placeholder, Describe Code...\nsubmission[\"Predicted\"] = bgm_predictions\nsubmission.to_csv(\"submission_iteration.csv\", index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}