{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Tabular 2022 - July Challenge","metadata":{}},{"cell_type":"code","source":"!pip install -qq sklego\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style('darkgrid')\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, MinMaxScaler, PowerTransformer\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import silhouette_score, calinski_harabasz_score, davies_bouldin_score\n\nfrom lightgbm import LGBMClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.discriminant_analysis import QuadraticDiscriminantAnalysis, LinearDiscriminantAnalysis\nfrom sklego.mixture import BayesianGMMClassifier","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\ndf = df.drop(columns=\"id\")\nsubmission = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Power Transformation","metadata":{}},{"cell_type":"code","source":"transformer = PowerTransformer()\nX_scaled = transformer.fit_transform(df)\nX_scaled = pd.DataFrame(X_scaled, columns = df.columns)\nbest_cols = ['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Soft Voting","metadata":{}},{"cell_type":"code","source":"def soft_voting(predict_number, best_cols = best_cols):\n    #initialise dataframe with 0's\n    predicted_probabilities = pd.DataFrame(np.zeros((len(df),7)), columns=range(1,8))\n    scores = []\n    # loop with a different random seeds\n    for i in tqdm(range(predict_number)):\n        #print(\"=========\", i, \"==========\")\n        X_scaled_sample = X_scaled.sample(66666)\n        gmm = BayesianGaussianMixture(n_components=7, covariance_type='full', max_iter=400, init_params=\"kmeans\", n_init=3, random_state=42*i)\n        gmm.fit(X_scaled_sample[best_cols])\n        pred_probs = gmm.predict_proba(X_scaled[best_cols])\n        pred_probs = pd.DataFrame(pred_probs, columns=range(1,8))\n        \n        # ensuring clusters are labeled the same value at each fit\n        if i == 0:\n            initial_centers = gmm.means_\n        new_classes = []\n        for mean2 in gmm.means_:\n            #for the current center of the current gmm, find the distances to every center in the initial gmm\n            distances = [np.linalg.norm(mean1-mean2) for mean1 in initial_centers]\n            # select the class with the minimum distance\n            new_class = np.argmin(distances) + 1 #add 1 as our labels are 1-7 but index is 0-6\n            new_classes.append(new_class)\n        # if the mapping from old cluster labels to new cluster labels isn't 1 to 1\n        if len(new_classes) != len(set(new_classes)):\n            print(\"iteration\", i, \"could not determine the cluster label mapping, skipping\")\n            continue\n        #apply the mapping by renaming the dataframe columns representing the original labels to the new labels    \n        pred_probs = pred_probs.rename(columns=dict(zip(range(1,8),new_classes)))\n        \n        #add the current prediction probabilities to the overall prediction probabilities\n        predicted_probabilities = predicted_probabilities + pred_probs\n        # lets score the cluster labels each iteration to see if soft voting is helpful\n        # db, ch, s = score_clusters(X_scaled[best_cols], predicted_probabilities.idxmax(axis=1), verbose=False)\n        # scores.append((db,ch,s))\n    \n    #normalise dataframe so each row sums to 1\n    predicted_probabilities = predicted_probabilities.div(predicted_probabilities.sum(axis=1), axis=0)\n    # display(pd.DataFrame(scores, columns=[\"Davies-Bouldin Index\",\"Calinski-Harabasz Index\",\"Silhouette Coefficient\"]))\n    return predicted_probabilities","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npred_probs = soft_voting(20)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create dataframe of predictions","metadata":{}},{"cell_type":"code","source":"def best_class(df):\n    new_df = df.copy()\n    new_df[\"highest_prob\"] = df.max(axis=1)\n    new_df[\"best_class\"] = df.idxmax(axis=1)\n    new_df[\"second_highest_prob\"] = df.apply(lambda x: x.nlargest(2).values[-1], axis=1)\n    new_df[\"second_best_class\"] = df.apply(lambda x: np.where(x == x.nlargest(2).values[-1])[0][0]+1, axis=1)\n    return new_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cluster_class_probs = best_class(pred_probs)\ncluster_class_probs.sample(10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Select only the clusters with higher prob than 0.8","metadata":{}},{"cell_type":"code","source":"confident_predictions = cluster_class_probs.loc[cluster_class_probs[\"highest_prob\"] >= 0.8]\nconfident_predictions_class = confident_predictions[\"best_class\"]\nX_scaled[\"class\"] = confident_predictions_class","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare the new Dataframes for Ensemble","metadata":{}},{"cell_type":"code","source":"train_df = X_scaled.loc[X_scaled[\"class\"] == X_scaled[\"class\"]]\ntest_df = X_scaled.loc[X_scaled[\"class\"] != X_scaled[\"class\"]]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(columns=\"class\").reset_index(drop=True)\ny = train_df[\"class\"].reset_index(drop=True)\nX_test = test_df.drop(columns=\"class\").reset_index(drop=True)\nX_full = X_scaled.drop(columns=\"class\")\nprint(f\"train_df: {len(X)}, test_df: {len(X_test)}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ensemble of models","metadata":{}},{"cell_type":"code","source":"model_et = ExtraTreesClassifier(n_estimators=2000, n_jobs=-1, random_state=42)\nmodel_lgbm = LGBMClassifier(objective='multiclass', n_estimators = 2000, random_state=42, learning_rate=0.05, n_jobs=-1)\nmodel_qda = QuadraticDiscriminantAnalysis()\nmodel_lda = LinearDiscriminantAnalysis()\nmodel_bgmm = BayesianGMMClassifier(n_components=7, random_state=1, tol=1e-3, covariance_type='full', max_iter=400, n_init=4, init_params='kmeans')\n\nmodels = {\"ET\":model_et, \"LGBM\":model_lgbm, \"QDA\":model_qda, \"LDA\":model_lda, \"BGMM_C\":model_bgmm}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_predict_all():\n    predictions = []\n    model_names = []\n    scores = []\n    for model_name, model in models.items():\n        print(\"===\",model_name,\"===\")\n        model.fit(X[best_cols], y)\n        preds_prob =  model.predict_proba(X_full[best_cols])\n        preds_prob_df = pd.DataFrame(preds_prob, columns=range(1,8), index=X_scaled.index)\n        # db, ch, s = score_clusters(X_scaled[best_cols], preds_prob_df.idxmax(axis=1), verbose=True)\n        # scores.append((db,ch,s))\n        predictions.append(preds_prob_df)\n        model_names.append(model_name)\n    \n    return predictions, model_names, scores","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions, model_names, scores = fit_predict_all()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Appending BGMM probabilities","metadata":{}},{"cell_type":"code","source":"cluster_class_probs = cluster_class_probs.loc[:,[1,2,3,4,5,6,7]]\npredictions.append(cluster_class_probs)\nmodel_names.append(\"BGMM\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Class weights probabilities","metadata":{}},{"cell_type":"code","source":"#chosen fairly randomly\npredictions_df = 0.5 * predictions[0] + 1.5 * predictions[1] + 0.5 * predictions[2] + 1.5 * predictions[4] + 0.5 * predictions[5]\n\n#normalise so rows sums to 1\npredictions_df = predictions_df.div(predictions_df.sum(axis=1), axis=0)\npredictions_df = best_class(predictions_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Update predictions using Bayesian GMM Classifier","metadata":{}},{"cell_type":"code","source":"def update_predictions(predict_number, y):\n    scores = []\n    for i in tqdm(range(predict_number)):\n        # print(\"=========\", i, \"==========\")\n        #X_scaled_sample = X_scaled.sample(66666)\n        if i==0:\n            X_scaled_sample = X_scaled[X_scaled['class'].notnull()]\n        else:\n            X_scaled_sample = X_scaled[X_scaled['pred_best_probs']>0.9]\n        y_sample = y.loc[X_scaled_sample.index]\n        print(len(X_scaled_sample))\n        bgmmC = BayesianGMMClassifier(\n        n_components=7,\n        random_state = i,\n        tol =1e-3,\n        covariance_type = 'full',\n        max_iter = 400,\n        n_init=3,\n        init_params='kmeans') # change to k-means++\n        \n        bgmmC.fit(X_scaled_sample[best_cols], y_sample)\n        \n        pred_probs = bgmmC.predict_proba(X_scaled[best_cols])\n        pred_probs = pd.DataFrame(pred_probs, columns=range(1,8))\n        X_scaled['pred_best_probs'] = pred_probs.max(axis=1)\n        \n        # lets score the cluster labels each iteration\n        # db, ch, s  = score_clusters(X_scaled[best_cols], pred_probs.idxmax(axis=1), verbose=False)\n        # scores.append((db, ch, s))\n        y = pred_probs.idxmax(axis=1)\n        \n    # display(pd.DataFrame(scores, columns=[\"Davies-Bouldin Index\",\"Calinski-Harabasz Index\",\"Silhouette Coefficient\"]))\n    return pred_probs","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_probabilities = update_predictions(predict_number=30, y=predictions_df[\"best_class\"])\npredictions_df = best_class(predicted_probabilities)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"submission[\"Predicted\"] = predictions_df[\"best_class\"]\nsubmission.to_csv('./submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}