{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Tabular July 22 Gaussian Mixture","metadata":{}},{"cell_type":"code","source":"!pip install scikit-lego\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, GridSearchCV, TimeSeriesSplit\nfrom sklearn.preprocessing import RobustScaler, PowerTransformer\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklego.mixture import BayesianGMMClassifier, GMMClassifier\nfrom sklearn.metrics import accuracy_score\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\nsub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_scaled = pd.DataFrame(PowerTransformer().fit_transform(df), columns=df.columns)\ndata_scaled = pd.DataFrame(\n    RobustScaler().fit_transform(data_scaled), columns=data_scaled.columns\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_cols = [\n    \"f_07\",\n    \"f_08\",\n    \"f_09\",\n    \"f_10\",\n    \"f_11\",\n    \"f_12\",\n    \"f_13\",\n    \"f_22\",\n    \"f_23\",\n    \"f_24\",\n    \"f_25\",\n    \"f_26\",\n    \"f_27\",\n    \"f_28\",\n]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = data_scaled[selected_cols].copy()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bayesian Gaussian Mixture","metadata":{}},{"cell_type":"code","source":"bgm = BayesianGaussianMixture(\n    n_components=7,\n    max_iter=300,\n    n_init=10,\n    random_state=2,\n    verbose_interval=100,\n)\n\nbgm_labels = bgm.fit_predict(data_scaled[selected_cols])\nbgm_proba = bgm.predict_proba(data_scaled[selected_cols])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Soft voting","metadata":{}},{"cell_type":"code","source":"n_components = 7\ndata_scaled[\"predict\"] = bgm_labels\ndata_scaled[\"predict_proba\"] = 0\n\nfor n in range(n_components):\n    data_scaled[f\"bgm_proba_{n}\"] = bgm_proba[:, n]\n    data_scaled.loc[data_scaled.predict == n, \"bgm_proba\"] = data_scaled[\n        f\"bgm_proba_{n}\"\n    ]\n\ntrain_index = np.array([])\nfor n in range(n_components):\n    median = data_scaled[data_scaled.predict == n][\"bgm_proba\"].median()\n\n    # Experiment with different thresholds\n    # Higher thereshold might overfit\n    n_inx = data_scaled[\n        (data_scaled.predict == n) & (data_scaled.bgm_proba > 0.675)\n    ].index\n\n    train_index = np.concatenate((train_index, n_inx))\n    print(\n        f\"class:{n}\",\n        f\"median: {round(median,4)}\",\n        \"Training data:\"\n        + str(round(len(n_inx) / len(data_scaled[(data_scaled.predict == n)]), 2) * 100)\n        + \"%\",\n    )\n\n\nprint(f\"\\nSize of Training data : {len(train_index)}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bayesian GMM Classifier","metadata":{}},{"cell_type":"code","source":"X = data_scaled.loc[train_index][selected_cols]\ny = data_scaled.loc[train_index][\"predict\"]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm = BayesianGMMClassifier(\n    n_components=7,\n    random_state=42,\n    # tol =1e-3,\n    covariance_type=\"full\",\n    max_iter=500,\n    n_init=7,\n    init_params=\"kmeans\", # you can use k-means++\n)\n\nbgm.fit(X, y)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"predictions = bgm.predict(test_data)\nsub[\"Predicted\"] = predictions\nsub.to_csv(\n    \"submission.csv\",\n    index=False,\n)","metadata":{},"execution_count":null,"outputs":[]}]}