{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc,random,os\nfrom sklearn.preprocessing import PowerTransformer,MinMaxScaler\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:30:32.887325Z","iopub.execute_input":"2022-07-19T13:30:32.888530Z","iopub.status.idle":"2022-07-19T13:30:34.340834Z","shell.execute_reply.started":"2022-07-19T13:30:32.888403Z","shell.execute_reply":"2022-07-19T13:30:34.339526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=714):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n\nseed=714\nseed_everything(seed)\n\nN_cluster=8","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:57:27.017888Z","iopub.execute_input":"2022-07-19T13:57:27.019012Z","iopub.status.idle":"2022-07-19T13:57:27.025209Z","shell.execute_reply.started":"2022-07-19T13:57:27.018966Z","shell.execute_reply":"2022-07-19T13:57:27.024032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\ntrain=train.drop(\"id\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:57:27.986985Z","iopub.execute_input":"2022-07-19T13:57:27.987802Z","iopub.status.idle":"2022-07-19T13:57:28.872189Z","shell.execute_reply.started":"2022-07-19T13:57:27.987766Z","shell.execute_reply":"2022-07-19T13:57:28.870913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#figured out which features I like based on running ANOVA\n#first ran ANOVA of int columns against continuous columns, like other discussions also mention\n#also ran ANOVA of int columns against interaction columns -- something like 800 columns and picked the significant ones\n\nanova_good_features=['f_22','f_23','f_24','f_25','f_26','f_27','f_28','f_07','f_08','f_09','f_10','f_11','f_12','f_13']\nanova_good_cross_features=['f_16','f_20','f_00xf_05','f_00xf_16','f_00xf_19','f_00xf_20','f_01xf_06','f_01xf_19','f_01xf_21','f_02xf_05','f_03xf_16','f_03xf_21','f_05xf_00','f_05xf_02','f_06xf_01','f_14xf_16','f_15xf_16','f_16xf_00','f_16xf_03','f_16xf_14','f_16xf_15','f_19xf_00','f_19xf_01','f_19xf_19','f_19xf_21','f_20xf_00','f_21xf_01','f_21xf_03','f_21xf_19','f_03xf_04','f_03xf_05','f_03xf_16','f_03xf_21','f_04xf_03','f_04xf_21','f_05xf_03','f_06xf_18','f_15xf_21','f_16xf_03','f_18xf_06','f_21xf_03','f_21xf_04','f_21xf_15','f_19','f_00xf_01','f_01xf_00','f_02xf_15','f_02xf_18','f_03xf_20','f_05xf_05','f_05xf_20','f_05xf_21','f_15xf_02','f_18xf_02','f_20xf_03','f_20xf_05','f_21xf_05','f_00xf_03','f_00xf_15','f_02xf_04','f_02xf_15','f_02xf_17','f_03xf_00','f_04xf_02','f_06xf_14','f_14xf_06','f_14xf_20','f_14xf_21','f_15xf_00','f_15xf_02','f_15xf_15','f_15xf_20','f_17xf_02','f_20xf_14','f_20xf_15','f_21xf_14','f_00xf_03','f_01xf_15','f_03xf_00','f_04xf_05','f_05xf_04','f_06xf_14','f_14xf_06','f_15xf_01','f_15xf_17','f_16xf_21','f_17xf_15','f_17xf_19','f_17xf_20','f_19xf_17','f_20xf_17','f_21xf_16','f_00xf_01','f_01xf_00','f_01xf_01','f_01xf_20','f_02xf_05','f_02xf_21','f_04xf_05','f_05xf_02','f_05xf_04','f_15xf_16','f_16xf_15','f_20xf_01','f_21xf_02','f_03xf_03','f_04xf_21','f_14xf_16','f_16xf_14','f_18xf_21','f_21xf_04','f_21xf_18']\n\n#clean up the cross features list which is repetitive and a bit of a mess\n#cross feature list has some duplicates, so get rid of them\n#make sure f_00xf_01 is the same as f_01xf00 and only keep one of those\n\nsingle_feat = [x for x in anova_good_cross_features if len(x) < 7]\ncross_feat = list(set([x for x in anova_good_cross_features if len(x) > 7]))\ncross_feat_sets = [set(x.split('x')) for x in cross_feat]\n\n\nseen = []\nfor i in cross_feat_sets:\n    if i in seen:\n        pass\n    else:\n        seen.append(i)\n\nfinal_cross = []\n\n#check for features that are squared vs multiplying two features\n#a set of a squared feature only gives one item rather than 2\nfor s in seen:\n    s1 = list(s)\n    a = s1[0]\n    if len(s1) == 1:\n        b = s1[0]\n    else:\n        b=s1[1]\n    final_cross.append(a+'x'+b)\n\n#finalize the list of columns to keep\nkeep_cols = anova_good_features.copy()\nkeep_cols.extend(single_feat)\nkeep_cols.extend(final_cross)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:57:29.726874Z","iopub.execute_input":"2022-07-19T13:57:29.727282Z","iopub.status.idle":"2022-07-19T13:57:29.744414Z","shell.execute_reply.started":"2022-07-19T13:57:29.727250Z","shell.execute_reply":"2022-07-19T13:57:29.743375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create interaction features\n#scale dataframe and only keep the 'keep_cols' I care about\n\ntrain_scaled=train.copy()\ncols = train.columns.to_list()\nfor c1 in cols:\n    for c2 in cols:\n        train_scaled[str(c1)+'x'+str(c2)] = train_scaled[c1] * train_scaled[c2]\n\n#just keep the columns I care about        \ntrain_scaled = train_scaled[keep_cols].copy()\n\n\n#transform columns. I think this messes everything up?\n\ntrain_scaled = MinMaxScaler().fit_transform(train_scaled)\ntrain_scaled = PowerTransformer().fit_transform(train_scaled)\ntrain_scaled = pd.DataFrame(train_scaled, columns=keep_cols)\n\nprint(type(train_scaled))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:57:45.017560Z","iopub.execute_input":"2022-07-19T13:57:45.017976Z","iopub.status.idle":"2022-07-19T13:57:55.873545Z","shell.execute_reply.started":"2022-07-19T13:57:45.017942Z","shell.execute_reply":"2022-07-19T13:57:55.872292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_scaled.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:57:56.743754Z","iopub.execute_input":"2022-07-19T13:57:56.744499Z","iopub.status.idle":"2022-07-19T13:57:56.771894Z","shell.execute_reply.started":"2022-07-19T13:57:56.744450Z","shell.execute_reply":"2022-07-19T13:57:56.770813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Make BGM model. \nIf using all columns, score ends up at zero. \nIf using original subset of columns, score is 0.58**","metadata":{}},{"cell_type":"code","source":"#\n# Try BayesianGaussianMixture\n#\n\nfrom sklearn.mixture import BayesianGaussianMixture\n\n#USE ONE OR THE OTHER:\n#model_cols = anova_good_features  #subset with no interaction terms -- gives a score of 0.58\nmodel_cols = train_scaled.columns.to_list()  #use all columns\n\n#model initialization\nbgm = BayesianGaussianMixture(n_components=N_cluster, \n                              covariance_type='full', \n                              tol=0.001, \n                              reg_covar=1e-06, \n                              max_iter=10000, \n                              n_init=1, \n                              init_params='kmeans', \n                              weight_concentration_prior_type='dirichlet_process', \n                              weight_concentration_prior=0.1, \n                              mean_precision_prior=None, \n                              mean_prior=None, \n                              degrees_of_freedom_prior=None, \n                              covariance_prior=None,\n                              random_state=seed, \n                              warm_start=False, verbose=2, \n                              verbose_interval=20)\n\nprint('Fitting model with ', len(model_cols), ' features.')\nbgm_results = bgm.fit_predict(train_scaled[model_cols])\nbgm_probs = bgm.predict_proba(train_scaled[model_cols])\n\n#get probabilities and prediction and put them in a dataframe\ncluster_cols = [str(x) for x in range(0, bgm_probs.shape[1])]\nbgm_probs_df = pd.DataFrame(bgm_probs, columns=cluster_cols)\nbgm_probs_df['max_prob'] = bgm_probs_df[cluster_cols].max(axis=1)\n\n#print(bgm_probs_df.head())\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:59:07.696270Z","iopub.execute_input":"2022-07-19T13:59:07.696740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"good_fits = pd.concat([train_scaled\n                       , pd.DataFrame(bgm_results, columns=['pred_c'])\n                       , bgm_probs_df], axis=1)\n#good_fits.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:52:27.737557Z","iopub.execute_input":"2022-07-19T13:52:27.737943Z","iopub.status.idle":"2022-07-19T13:52:27.768094Z","shell.execute_reply.started":"2022-07-19T13:52:27.737912Z","shell.execute_reply":"2022-07-19T13:52:27.766890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#the more cluster predictions with high probability hopefully means a better model\n\n#histogram showing the probability that led to the cluster group prediction.\n#This uses 8 clusters, so the minimum percentage would be 100% / 8 = 12.5%\n#But we want a good model, so the percentages should be higher indicating a stronger confidence\n\ngood_fits['max_prob'].hist(bins=100)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:52:29.679011Z","iopub.execute_input":"2022-07-19T13:52:29.679429Z","iopub.status.idle":"2022-07-19T13:52:30.062374Z","shell.execute_reply.started":"2022-07-19T13:52:29.679393Z","shell.execute_reply":"2022-07-19T13:52:30.061165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save the predictions for submission\npreds = good_fits['pred_c'].copy()\npreds.index.name='Id'\npreds.name='Predicted'\npreds.to_csv('./preds_071922_gbm_all.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:52:34.917957Z","iopub.execute_input":"2022-07-19T13:52:34.918676Z","iopub.status.idle":"2022-07-19T13:52:35.120099Z","shell.execute_reply.started":"2022-07-19T13:52:34.918640Z","shell.execute_reply":"2022-07-19T13:52:35.119057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}